{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### make w2v feature example","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport gensim\nfrom gensim.models import word2vec","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-17T02:25:28.394020Z","iopub.execute_input":"2022-11-17T02:25:28.394387Z","iopub.status.idle":"2022-11-17T02:25:28.399943Z","shell.execute_reply.started":"2022-11-17T02:25:28.394360Z","shell.execute_reply":"2022-11-17T02:25:28.398791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cite = pd.read_hdf('../input/open-problems-multimodal/train_cite_inputs.h5')\n\n# this notebook use small sample. Essentially, I use all data, including test.\ncite = cite.head(1000)\ncite.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-11-17T02:25:28.497675Z","iopub.execute_input":"2022-11-17T02:25:28.498003Z","iopub.status.idle":"2022-11-17T02:26:06.849168Z","shell.execute_reply.started":"2022-11-17T02:25:28.497978Z","shell.execute_reply":"2022-11-17T02:26:06.848272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rank = 100\nw2v_dim = 16\n\n# get order\nsort_array = np.argsort(-np.array(cite))[:,0:rank].astype('str')\n\n# from array to list to train word2vec\nsort_list = []\nfor i in sort_array:\n    sort_list.append(list(i))\n    \n# get gene vectors\nmodel = word2vec.Word2Vec(sentences=sort_list, vector_size=w2v_dim, window=5, min_count=1)\nword_vectors = model.wv","metadata":{"execution":{"iopub.status.busy":"2022-11-17T02:26:06.850949Z","iopub.execute_input":"2022-11-17T02:26:06.851294Z","iopub.status.idle":"2022-11-17T02:26:07.699885Z","shell.execute_reply.started":"2022-11-17T02:26:06.851261Z","shell.execute_reply":"2022-11-17T02:26:07.699038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# example: gene index 13374 16dim vector\nword_vectors['13374']","metadata":{"execution":{"iopub.status.busy":"2022-11-17T02:26:07.700957Z","iopub.execute_input":"2022-11-17T02:26:07.701246Z","iopub.status.idle":"2022-11-17T02:26:07.707707Z","shell.execute_reply.started":"2022-11-17T02:26:07.701215Z","shell.execute_reply":"2022-11-17T02:26:07.706844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get average of the top 100 ranked gene vectors per cell\n\nword_vec_list = []\n\nfor l in range(len(sort_array)):\n    \n    word_vec =np.zeros([w2v_dim])\n    \n    for i, j in enumerate(sort_array[l]):\n        word_vec += word_vectors[j] / len(sort_array[l])    \n    \n    word_vec_list.append(word_vec)\n    \nw2v_feature = pd.DataFrame(word_vec_list).add_prefix('w2v_')","metadata":{"execution":{"iopub.status.busy":"2022-11-17T02:26:07.709408Z","iopub.execute_input":"2022-11-17T02:26:07.709741Z","iopub.status.idle":"2022-11-17T02:26:08.183681Z","shell.execute_reply.started":"2022-11-17T02:26:07.709717Z","shell.execute_reply":"2022-11-17T02:26:08.182821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"w2v_feature.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-11-17T02:26:08.184704Z","iopub.execute_input":"2022-11-17T02:26:08.184973Z","iopub.status.idle":"2022-11-17T02:26:08.204878Z","shell.execute_reply.started":"2022-11-17T02:26:08.184950Z","shell.execute_reply":"2022-11-17T02:26:08.203816Z"},"trusted":true},"execution_count":null,"outputs":[]}]}