{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":38760,"databundleVersionId":4493939,"sourceType":"competition"},{"sourceId":4474043,"sourceType":"datasetVersion","datasetId":2601572}],"dockerImageVersionId":30357,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Data preprocessing","metadata":{}},{"cell_type":"code","source":"!pip install polars\n","metadata":{"execution":{"iopub.status.busy":"2023-12-22T14:34:37.088763Z","iopub.execute_input":"2023-12-22T14:34:37.089465Z","iopub.status.idle":"2023-12-22T14:34:52.475783Z","shell.execute_reply.started":"2023-12-22T14:34:37.089421Z","shell.execute_reply":"2023-12-22T14:34:52.474408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\n","metadata":{"execution":{"iopub.status.busy":"2023-12-22T14:34:52.478152Z","iopub.execute_input":"2023-12-22T14:34:52.478763Z","iopub.status.idle":"2023-12-22T14:34:52.486682Z","shell.execute_reply.started":"2023-12-22T14:34:52.478700Z","shell.execute_reply":"2023-12-22T14:34:52.484760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import polars as pl\n","metadata":{"execution":{"iopub.status.busy":"2023-12-22T14:34:52.488740Z","iopub.execute_input":"2023-12-22T14:34:52.489197Z","iopub.status.idle":"2023-12-22T14:34:52.501080Z","shell.execute_reply.started":"2023-12-22T14:34:52.489157Z","shell.execute_reply":"2023-12-22T14:34:52.499035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from gensim.test.utils import common_texts\nfrom gensim.models import Word2Vec","metadata":{"execution":{"iopub.status.busy":"2023-12-22T14:34:52.503512Z","iopub.execute_input":"2023-12-22T14:34:52.503978Z","iopub.status.idle":"2023-12-22T14:34:52.515705Z","shell.execute_reply.started":"2023-12-22T14:34:52.503912Z","shell.execute_reply":"2023-12-22T14:34:52.514250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pl.read_parquet('../input/otto-full-optimized-memory-footprint/train.parquet')\ntest = pl.read_parquet('../input/otto-full-optimized-memory-footprint/test.parquet')","metadata":{"execution":{"iopub.status.busy":"2023-12-22T14:34:52.518411Z","iopub.execute_input":"2023-12-22T14:34:52.518866Z","iopub.status.idle":"2023-12-22T14:35:01.849325Z","shell.execute_reply.started":"2023-12-22T14:34:52.518826Z","shell.execute_reply":"2023-12-22T14:35:01.847937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sentences_df =  pl.concat([train, test]).groupby('session').agg(\n    pl.col('aid').alias('sentence')\n)\n\nsentences = sentences_df['sentence'].to_list()\ndel sentences_df; gc.collect() ","metadata":{"execution":{"iopub.status.busy":"2023-12-22T14:35:01.856075Z","iopub.execute_input":"2023-12-22T14:35:01.857326Z","iopub.status.idle":"2023-12-22T14:36:11.319948Z","shell.execute_reply.started":"2023-12-22T14:35:01.857219Z","shell.execute_reply":"2023-12-22T14:36:11.318517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training word2vec model","metadata":{}},{"cell_type":"code","source":"%%time\n\nw2vec = Word2Vec(sentences=sentences, vector_size= 64, window = 3, negative = 8, ns_exponent = 0.2, sg = 1, min_count=1, workers=4)","metadata":{"execution":{"iopub.status.busy":"2023-12-22T14:36:11.322143Z","iopub.execute_input":"2023-12-22T14:36:11.323145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preparing aproximated nearest neighbour model","metadata":{}},{"cell_type":"code","source":"%%time\n\nfrom annoy import AnnoyIndex\n\naid2idx = {aid: i for i, aid in enumerate(w2vec.wv.index_to_key)}\nindex = AnnoyIndex(64, 'euclidean')\n\nfor aid, idx in aid2idx.items():\n    index.add_item(idx, w2vec.wv.vectors[idx])\n    \nindex.build(32)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating aids predictions","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nfrom collections import defaultdict\nimport collections\n\nsession_types = ['clicks', 'carts', 'orders']\ntest_session_AIDs = test.to_pandas().reset_index(drop=True).groupby('session')['aid'].apply(list)\ntest_session_types = test.to_pandas().reset_index(drop=True).groupby('session')['type'].apply(list)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = []\n\ntype_weight_multipliers = {0: 1, 1: 6, 2: 3}\n\nsession_num = len(test_session_AIDs)\n\nfor AIDs, types in zip(test_session_AIDs[:session_num], test_session_types[:session_num]):\n    if len(AIDs) >= 20:\n        # if we have enough aids (over equals 20) we don't need to look for candidates! we just use the old logic\n        weights=np.logspace(0.1,1,len(AIDs),base=2, endpoint=True)-1\n        aids_temp=defaultdict(lambda: 0)\n        for aid,w,t in zip(AIDs,weights,types): \n            aids_temp[aid]+= w * type_weight_multipliers[t]\n            \n        sorted_aids=[k for k, v in sorted(aids_temp.items(), key=lambda item: -item[1])]\n        labels.append(sorted_aids[:20])\n    else:\n        # here we don't have 20 aids to output -- we will use word2vec embeddings to generate candidates!\n        AIDs = list(dict.fromkeys(AIDs[::-1]))\n        \n        # let's grab the up to 3 recent aids\n        recent_len = max(min(3,len(AIDs)),1)\n        \n        # how many aids for each aid\n        AIDs_num = round((20-len(AIDs))/recent_len) + 2\n        \n        # let's look for some neighbors!        \n        nns_it = []\n        for it in range(0,recent_len):\n            nns_it += [w2vec.wv.index_to_key[i] for i in index.get_nns_by_item(aid2idx[AIDs[it]], AIDs_num)[1:]]\n        \n        # select repeating and unique neighbors\n        nns_repeated = [item for item, count in collections.Counter(nns_it).items() if count > 1]\n        nns_once = [item for item, count in collections.Counter(nns_it).items() if count == 1]\n\n        # prepare selection\n        nns = (nns_repeated+nns_once)[:20]\n        labels.append((AIDs+nns)[:20])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preparing submission dataframe","metadata":{}},{"cell_type":"code","source":"labels_as_strings = [' '.join([str(l) for l in lls]) for lls in labels]\n\npredictions = pd.DataFrame(data={'session_type': test_session_AIDs.index, 'labels': labels_as_strings})\n\nprediction_dfs = []\n\nfor st in session_types:\n    modified_predictions = predictions.copy()\n    modified_predictions.session_type = modified_predictions.session_type.astype('str') + f'_{st}'\n    prediction_dfs.append(modified_predictions)\n\nsubmission = pd.concat(prediction_dfs).reset_index(drop=True)\nsubmission.to_csv('submission.csv', index=False)\n\ndel labels, labels_as_strings, predictions, prediction_dfs\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":" ","metadata":{}}]}