{"cells":[{"metadata":{"trusted":true,"_uuid":"e3b964a8d83090bca25b72a06c1dfa9d65ce2154"},"cell_type":"code","source":"import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport numpy as np\n\ntrain_df = pd.read_csv('../input/train.csv')\ntest_df = pd.read_csv('../input/test.csv')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"80131c1e13f4bd2675ec1af2e0c7554a1e149a9c"},"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nvectorizer = CountVectorizer()\nvectorizer.fit(train_df.question_text.values)\n\ntfidf=vectorizer.fit_transform(train_df.question_text.values)\nwoord=vectorizer.get_feature_names()\n\nvectorizer.fit_transform(train_df[train_df.target==0].question_text.values)\nwoord0=vectorizer.get_feature_names()\n\nvectorizer.fit_transform(train_df[train_df.target==1].question_text.values)\nwoord1=vectorizer.get_feature_names()\n\ndef intersection(lst1, lst2): \n    lst3 = [value for value in lst1 if value in lst2] \n    return lst3 \nwoord01=intersection(set(woord0),set(woord1))\nprint(len(woord01),len(woord1),len(woord0))\ndef exclinters(lst1, lst2): \n    lst3 = [value for value in lst1 if value not in lst2] \n    return lst3 \nwordsimpf=exclinters(set(woord1),set(woord01))\nprint(len(woord),len(wordsimpf))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f291f5c72f2059d0307ae3f2c0cb5f9ed6f8535c"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembedding_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE)  )\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8f1f8f78be7402eca3ff0f8c8a57b81a8a0e705b","scrolled":true},"cell_type":"code","source":"glove=pd.DataFrame([] )\nfor xi in range(int(len(woord)/1000)+1):\n    \n    emb =pd.DataFrame([] )\n    for word in woord[xi*1000:xi*1000+1000]:\n        emb=emb.append(pd.DataFrame(embedding_index.get(word),columns=[word] ).T)\n    glove=glove.append(emb)    \n                    \n        #print(emb.shape)\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"af83dc66aaba85b2fa2dbc89b56593509ca0eb6f"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembedding_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE) if len(o)>100)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9b1009c6bba611252617a1104feae15e26a345f8"},"cell_type":"code","source":"wiki=pd.DataFrame([] )\nfor xi in range(int(len(woord)/1000)+1):\n    \n    emb =pd.DataFrame([] )\n    for word in woord[xi*1000:xi*1000+1000]:\n        emb=emb.append(pd.DataFrame(embedding_index.get(word),columns=[word] ).T)\n    wiki=wiki.append(emb)    \n                    \n        #print(emb.shape)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"31d3bf0345ec4bc91397ad60235b64563531a8f3"},"cell_type":"code","source":"wiki","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5a4998dd19db915c35701b03e4b54d470cf94fcf"},"cell_type":"code","source":"del EMBEDDING_FILE\ndel embedding_index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"43b8f4c46deb46e09ddce2289caf58bce66a5224"},"cell_type":"code","source":"wiki=wiki.dropna()\nglove=glove.dropna()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c387287fc974d90e193ed4bdfd46a78f12979b0b"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1ee9ee2f469d84c0d63db71780098c22c9b8416d"},"cell_type":"code","source":"# from https://stackoverflow.com/questions/21030391/how-to-normalize-array-numpy\ndef normalized(a, axis=-1, order=2):\n    \"\"\"Utility function to normalize the rows of a numpy array.\"\"\"\n    l2 = np.atleast_1d(np.linalg.norm(a, order, axis))\n    l2[l2==0] = 1\n    return a / np.expand_dims(l2, axis)\n\ndef make_training_matrices(source_dictionary, target_dictionary, bilingual_dictionary):\n    \"\"\"\n    Source and target dictionaries are the FastVector objects of\n    source/target languages. bilingual_dictionary is a list of \n    translation pair tuples [(source_word, target_word), ...].\n    \"\"\"\n    source_matrix = pd.DataFrame([])\n    target_matrix = pd.DataFrame([])\n\n    for (source, target) in bilingual_dictionary:\n        if source in source_dictionary and target in target_dictionary:\n            source_matrix.append(source_dictionary.loc[source])\n            target_matrix.append(target_dictionary.loc[target])\n\n    # return training matrices\n    return source_matrix, target_matrix\n\ndef learn_transformation(source_matrix, target_matrix, normalize_vectors=True):\n    \"\"\"\n    Source and target matrices are numpy arrays, shape\n    (dictionary_length, embedding_dimension). These contain paired\n    word vectors from the bilingual dictionary.\n    \"\"\"\n    # optionally normalize the training vectors\n    if normalize_vectors:\n        source_matrix = normalized(source_matrix)\n        target_matrix = normalized(target_matrix)\n\n    # perform the SVD\n    product = np.matmul(source_matrix.transpose(), target_matrix)\n    print(product)\n    U, s, V = np.linalg.svd(product)\n\n    # return orthogonal transformation which aligns source language to the target\n    return np.matmul(U, V)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f8ffa4ba8c9756872f041f6f0815c6192f2225de","scrolled":false},"cell_type":"code","source":"\noverlap = list(wiki.index & glove.index)\nbilingual_dictionary = [(entry, entry) for entry in overlap]\n\n# form the training matrices\nsource_matrix=wiki.loc[overlap]\ntarget_matrix=glove.loc[overlap]\nprint( source_matrix.shape,target_matrix.shape )\n\n# learn and apply the transformation\ntransform = learn_transformation(source_matrix.values, target_matrix.values)\nuniform=np.dot( wiki.values,transform )\nuniform","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7f907621c448f71c84444bdabafcad523ee530be"},"cell_type":"code","source":"uniform=pd.DataFrame(uniform,index=wiki.index)\nuniform.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8b9cc13f3a25577ffad70fb676f9b8d934ee843b"},"cell_type":"code","source":"from sklearn.metrics.pairwise import euclidean_distances,cosine_similarity\nprint( cosine_similarity(wiki.loc[['cat','dog']],wiki.loc[['cat','dog']]) )\nprint( cosine_similarity(wiki.loc[['cat','dog']],glove.loc[['cat','dog']]) )\n\n#the uniform matrix is the wiki database transformed to the glove\nprint( cosine_similarity(uniform.loc[['cat','dog']],glove.loc[['cat','dog']]) )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"629d48a3f65d9a563be847828c05503bbad86b95"},"cell_type":"code","source":"# form the training matrices\nsource_matrix=glove.loc[overlap]\ntarget_matrix=wiki.loc[overlap]\nprint( source_matrix.shape,target_matrix.shape )\n\n# learn and apply the transformation\ntransform = learn_transformation(source_matrix.values, target_matrix.values)\nuniform=np.dot( glove.values,transform )\nuniform","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"83b9ea0197684bd8b265db33ce55b1711bff91e1"},"cell_type":"code","source":"uniform=pd.DataFrame(uniform,index=glove.index)\nuniform.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"445bfb20ecbd2e8cb851f33a296c9a70a1bac2b5"},"cell_type":"code","source":"from sklearn.metrics.pairwise import euclidean_distances,cosine_similarity\nprint( cosine_similarity(wiki.loc[['cat','dog']],wiki.loc[['cat','dog']]) )\nprint( cosine_similarity(wiki.loc[['cat','dog']],glove.loc[['cat','dog']]) )\n\n#the uniform matrix is the gloe database transformed to the wiki\nprint( cosine_similarity(uniform.loc[['cat','dog']],wiki.loc[['cat','dog']]) )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bb35d255f4e624617054234fa7f87db8b3172f69"},"cell_type":"code","source":"glove.to_csv('glove.csv')\nwiki.to_csv('wiki.csv')\nuniform.to_csv('uniform.csv')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}