{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In this notebook we will train a <code>Word2Vec</code> model, aproximated nearest neighbour model and test it on local validation set.\n\nWe will use the <code>gensim</code> library for <code>Word2Vec</code>, aproximated nearest neighbour model form <code>annoy</code>.\n\nFor this purpose, I used the [dataset](https://www.kaggle.com/datasets/radek1/otto-full-optimized-memory-footprint) provided by Radek Osmulski. The prepared sets allow you to test the developed recommendation system. The developed code in this notebook is based on the example prepared by Radek (see [nootebook](https://www.kaggle.com/code/radek1/word2vec-how-to-training-and-submission)). To improve the result obtained by the model, we introduce some modifications. For tips related to these parameters please see [notebook](https://www.kaggle.com/code/balaganiarz0/word2vec-model-local-validation).","metadata":{}},{"cell_type":"markdown","source":"# Data preprocessing","metadata":{}},{"cell_type":"code","source":"!pip install polars\nimport gc\nimport polars as pl\nfrom gensim.test.utils import common_texts\nfrom gensim.models import Word2Vec\n\ntrain = pl.read_parquet('../input/otto-full-optimized-memory-footprint/train.parquet')\ntest = pl.read_parquet('../input/otto-full-optimized-memory-footprint/test.parquet')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sentences_df =  pl.concat([train, test]).groupby('session').agg(\n    pl.col('aid').alias('sentence')\n)\n\nsentences = sentences_df['sentence'].to_list()\ndel sentences_df; gc.collect() ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training word2vec model","metadata":{}},{"cell_type":"code","source":"%%time\n\nw2vec = Word2Vec(sentences=sentences, vector_size= 64, window = 3, negative = 8, ns_exponent = 0.2, sg = 1, min_count=1, workers=4)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preparing aproximated nearest neighbour model","metadata":{}},{"cell_type":"code","source":"%%time\n\nfrom annoy import AnnoyIndex\n\naid2idx = {aid: i for i, aid in enumerate(w2vec.wv.index_to_key)}\nindex = AnnoyIndex(64, 'euclidean')\n\nfor aid, idx in aid2idx.items():\n    index.add_item(idx, w2vec.wv.vectors[idx])\n    \nindex.build(32)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating aids predictions","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nfrom collections import defaultdict\nimport collections\n\nsession_types = ['clicks', 'carts', 'orders']\ntest_session_AIDs = test.to_pandas().reset_index(drop=True).groupby('session')['aid'].apply(list)\ntest_session_types = test.to_pandas().reset_index(drop=True).groupby('session')['type'].apply(list)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = []\n\ntype_weight_multipliers = {0: 1, 1: 6, 2: 3}\n\nsession_num = len(test_session_AIDs)\n\nfor AIDs, types in zip(test_session_AIDs[:session_num], test_session_types[:session_num]):\n    if len(AIDs) >= 20:\n        # if we have enough aids (over equals 20) we don't need to look for candidates! we just use the old logic\n        weights=np.logspace(0.1,1,len(AIDs),base=2, endpoint=True)-1\n        aids_temp=defaultdict(lambda: 0)\n        for aid,w,t in zip(AIDs,weights,types): \n            aids_temp[aid]+= w * type_weight_multipliers[t]\n            \n        sorted_aids=[k for k, v in sorted(aids_temp.items(), key=lambda item: -item[1])]\n        labels.append(sorted_aids[:20])\n    else:\n        # here we don't have 20 aids to output -- we will use word2vec embeddings to generate candidates!\n        AIDs = list(dict.fromkeys(AIDs[::-1]))\n        \n        # let's grab the up to 3 recent aids\n        recent_len = max(min(3,len(AIDs)),1)\n        \n        # how many aids for each aid\n        AIDs_num = round((20-len(AIDs))/recent_len) + 2\n        \n        # let's look for some neighbors!        \n        nns_it = []\n        for it in range(0,recent_len):\n            nns_it += [w2vec.wv.index_to_key[i] for i in index.get_nns_by_item(aid2idx[AIDs[it]], AIDs_num)[1:]]\n        \n        # select repeating and unique neighbors\n        nns_repeated = [item for item, count in collections.Counter(nns_it).items() if count > 1]\n        nns_once = [item for item, count in collections.Counter(nns_it).items() if count == 1]\n\n        # prepare selection\n        nns = (nns_repeated+nns_once)[:20]\n        labels.append((AIDs+nns)[:20])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preparing submission dataframe","metadata":{}},{"cell_type":"code","source":"labels_as_strings = [' '.join([str(l) for l in lls]) for lls in labels]\n\npredictions = pd.DataFrame(data={'session_type': test_session_AIDs.index, 'labels': labels_as_strings})\n\nprediction_dfs = []\n\nfor st in session_types:\n    modified_predictions = predictions.copy()\n    modified_predictions.session_type = modified_predictions.session_type.astype('str') + f'_{st}'\n    prediction_dfs.append(modified_predictions)\n\nsubmission = pd.concat(prediction_dfs).reset_index(drop=True)\nsubmission.to_csv('submission.csv', index=False)\n\ndel labels, labels_as_strings, predictions, prediction_dfs\ngc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Wishing you all the best in the challenge! 🤞","metadata":{}}]}