{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook is to train a `Word2Vec` model.\n\nWe will use the `gensim` library which offers extremely fast training on the CPU.\n\nWe will rely on `polars` and its small memory footprint to load and process the data. To speed things up, use “otto-ful-optimized-memory-footprint” dataset in a parquet format","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"# Data Preprocessing","metadata":{}},{"cell_type":"code","source":"!pip install polars\n\nimport polars as pl\nfrom gensim.test.utils import common_texts\nfrom gensim.models import Word2Vec\n\ntrain = pl.read_parquet('../input/otto-full-optimized-memory-footprint/train.parquet')\ntest = pl.read_parquet('../input/otto-full-optimized-memory-footprint/test.parquet')","metadata":{"execution":{"iopub.status.busy":"2023-01-05T05:04:39.578432Z","iopub.execute_input":"2023-01-05T05:04:39.579976Z","iopub.status.idle":"2023-01-05T05:05:19.775915Z","shell.execute_reply.started":"2023-01-05T05:04:39.579846Z","shell.execute_reply":"2023-01-05T05:05:19.775141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now transform the data into a format that the `gensim` library can work with. `polars` makes the process very efficiently and quickly.","metadata":{}},{"cell_type":"code","source":"sentences_df = pl.concat([train, test]).groupby('session').agg(\n    pl.col('aid').alias('sentence')\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-05T05:05:19.777512Z","iopub.execute_input":"2023-01-05T05:05:19.777770Z","iopub.status.idle":"2023-01-05T05:05:28.105012Z","shell.execute_reply.started":"2023-01-05T05:05:19.777746Z","shell.execute_reply":"2023-01-05T05:05:28.103742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sentences = sentences_df['sentence'].to_list()","metadata":{"execution":{"iopub.status.busy":"2023-01-05T05:05:28.106139Z","iopub.execute_input":"2023-01-05T05:05:28.106417Z","iopub.status.idle":"2023-01-05T05:05:58.292232Z","shell.execute_reply.started":"2023-01-05T05:05:28.106391Z","shell.execute_reply":"2023-01-05T05:05:58.291004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training a word2vec model","metadata":{}},{"cell_type":"code","source":"%%time\n\nw2vec = Word2Vec(sentences=sentences, vector_size=32, min_count=1, workers=4)","metadata":{"execution":{"iopub.status.busy":"2023-01-05T05:05:58.294988Z","iopub.execute_input":"2023-01-05T05:05:58.295326Z","iopub.status.idle":"2023-01-05T05:31:00.271122Z","shell.execute_reply.started":"2023-01-05T05:05:58.295297Z","shell.execute_reply":"2023-01-05T05:31:00.269863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"With the model fully train, let us use similarity between trained representations of our `aids` to create a submission.\n\nThe search functionality where we look for nearest neighbors in the embedding space is built into `gensim`, but it is unfortunately super slow. Let's use `annoy` which is much faster (it performs approximate nearest neigbor search).","metadata":{}},{"cell_type":"code","source":"%%time\n\nfrom annoy import AnnoyIndex\n\naid2idx = {aid: i for i, aid in enumerate(w2vec.wv.index_to_key)}\nindex = AnnoyIndex(32, 'euclidean')\n\nfor aid, idx in aid2idx.items():\n    index.add_item(idx, w2vec.wv.vectors[idx])\n    \nindex.build(10)","metadata":{"execution":{"iopub.status.busy":"2023-01-05T05:31:00.272570Z","iopub.execute_input":"2023-01-05T05:31:00.272882Z","iopub.status.idle":"2023-01-05T05:31:18.311518Z","shell.execute_reply.started":"2023-01-05T05:31:00.272854Z","shell.execute_reply":"2023-01-05T05:31:18.310350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Outputting a submission","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nfrom collections import defaultdict\n\nsample_sub = pd.read_csv('../input/otto-recommender-system//sample_submission.csv')\n\nsession_types = ['clicks', 'carts', 'orders']\ntest_session_AIDs = test.to_pandas().reset_index(drop=True).groupby('session')['aid'].apply(list)\ntest_session_types = test.to_pandas().reset_index(drop=True).groupby('session')['type'].apply(list)\n\nlabels = []\n\n# we use the same best weight for item type as we find in Tuning Candidate ReRank Model\n# (carts are of greatest importance)\ntype_weight_multipliers = {0: 0.5, 1: 9, 2: 0.5}\nfor AIDs, types in zip(test_session_AIDs, test_session_types):\n    if len(AIDs) >= 20:\n        # if we have enough aids (over equals 20) we don't need to look for candidates! we just use the old logic\n        weights=np.logspace(0.1,1,len(AIDs),base=2, endpoint=True)-1\n        aids_temp=defaultdict(lambda: 0)\n        for aid,w,t in zip(AIDs,weights,types): \n            aids_temp[aid]+= w * type_weight_multipliers[t]\n            \n        sorted_aids=[k for k, v in sorted(aids_temp.items(), key=lambda item: -item[1])]\n        labels.append(sorted_aids[:20])\n    else:\n        # here we don't have 20 aids to output -- we will use word2vec embeddings to generate candidates!\n        AIDs = list(dict.fromkeys(AIDs[::-1]))\n        \n        # grab the most recent aid\n        most_recent_aid = AIDs[0]\n        \n        # look for their nearest neighbors (besides oneself)\n        nns = [w2vec.wv.index_to_key[i] for i in index.get_nns_by_item(aid2idx[most_recent_aid], 21)[1:]]\n                        \n        labels.append((AIDs+nns)[:20])","metadata":{"execution":{"iopub.status.busy":"2023-01-05T05:31:18.313093Z","iopub.execute_input":"2023-01-05T05:31:18.313489Z","iopub.status.idle":"2023-01-05T05:34:03.356578Z","shell.execute_reply.started":"2023-01-05T05:31:18.313451Z","shell.execute_reply":"2023-01-05T05:34:03.355547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now pull it all together and write it to a file.","metadata":{}},{"cell_type":"code","source":"labels_as_strings = [' '.join([str(l) for l in lls]) for lls in labels]\n\npredictions = pd.DataFrame(data={'session_type': test_session_AIDs.index, 'labels': labels_as_strings})\n\nprediction_dfs = []\n\nfor st in session_types:\n    modified_predictions = predictions.copy()\n    modified_predictions.session_type = modified_predictions.session_type.astype('str') + f'_{st}'\n    prediction_dfs.append(modified_predictions)\n\nsubmission = pd.concat(prediction_dfs).reset_index(drop=True)\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-05T05:34:03.357660Z","iopub.execute_input":"2023-01-05T05:34:03.357963Z","iopub.status.idle":"2023-01-05T05:34:32.334626Z","shell.execute_reply.started":"2023-01-05T05:34:03.357936Z","shell.execute_reply":"2023-01-05T05:34:32.333291Z"},"trusted":true},"execution_count":null,"outputs":[]}]}