{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\n### what - the goal\nof this notebook is to do live plot word2vec loss.\n\n### why \nto debug the model.\n\n### how \nuse word2vec callback function.\n","metadata":{}},{"cell_type":"markdown","source":"# import libs needed","metadata":{}},{"cell_type":"code","source":"# live plot\n!pip install livelossplot \nfrom livelossplot import PlotLosses\n\nplotlosses = PlotLosses()","metadata":{"execution":{"iopub.status.busy":"2023-01-20T05:21:25.405177Z","iopub.execute_input":"2023-01-20T05:21:25.405669Z","iopub.status.idle":"2023-01-20T05:21:39.094346Z","shell.execute_reply.started":"2023-01-20T05:21:25.405630Z","shell.execute_reply":"2023-01-20T05:21:39.093084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# define the callback","metadata":{}},{"cell_type":"code","source":"from gensim.models.callbacks import CallbackAny2Vec\nimport os\nimport math\nimport sys\nfrom datetime import datetime\nimport psutil  \nimport numpy as np\n\ndef sys_stats():\n    pid = os.getpid()\n    ps = psutil.Process(pid)\n    memory_usage = ps.memory_info()[0] / 2. ** 30\n    return f'{datetime.now()}  memory usage GB:' + str(np.round(memory_usage, 2))\n\nclass w2vcallback(CallbackAny2Vec):\n    '''Callback to print loss after each epoch.'''\n  \n    def __init__(self):\n        self.epoch = 0\n        self.loss_last = 0\n\n    def on_epoch_end(self, model):\n        loss = model.get_latest_training_loss()\n        # define your own calculation of the loss\n        loss_now = loss - self.loss_last\n        self.loss_last = loss\n        self.epoch += 1\n        \n        print(f'{sys_stats()} Loss after epoch {self.epoch}: {loss_now}')   \n        plotlosses.update({    \n           'loss': loss_now, ### / (epoch + 2.),\n        })\n        plotlosses.send()\n","metadata":{"execution":{"iopub.status.busy":"2023-01-20T05:23:14.754217Z","iopub.execute_input":"2023-01-20T05:23:14.755135Z","iopub.status.idle":"2023-01-20T05:23:14.765705Z","shell.execute_reply.started":"2023-01-20T05:23:14.755085Z","shell.execute_reply":"2023-01-20T05:23:14.764466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install polars\nimport gc\nimport polars as pl\nfrom gensim.test.utils import common_texts\nfrom gensim.models import Word2Vec\n\ntrain = pl.read_parquet('../input/otto-full-optimized-memory-footprint/train.parquet')\ntest = pl.read_parquet('../input/otto-full-optimized-memory-footprint/test.parquet')","metadata":{"execution":{"iopub.status.busy":"2023-01-20T05:23:25.834199Z","iopub.execute_input":"2023-01-20T05:23:25.834700Z","iopub.status.idle":"2023-01-20T05:23:53.909052Z","shell.execute_reply.started":"2023-01-20T05:23:25.834665Z","shell.execute_reply":"2023-01-20T05:23:53.908021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sentences_df =  pl.concat([train, test]).groupby('session').agg(\n    pl.col('aid').alias('sentence')\n)\n\nsentences = sentences_df['sentence'].to_list()\ndel sentences_df; gc.collect() ","metadata":{"execution":{"iopub.status.busy":"2023-01-20T05:23:53.911592Z","iopub.execute_input":"2023-01-20T05:23:53.912874Z","iopub.status.idle":"2023-01-20T05:24:53.116612Z","shell.execute_reply.started":"2023-01-20T05:23:53.912814Z","shell.execute_reply":"2023-01-20T05:24:53.115370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# chain the callback to word2vec","metadata":{}},{"cell_type":"markdown","source":"make word2vec more reproducable","metadata":{}},{"cell_type":"code","source":"%env PYTHONHASHSEED=1\nimport hashlib\ndef hashf(s):\n    return int(hashlib.md5(str(s).encode()).hexdigest(), 32)","metadata":{"execution":{"iopub.status.busy":"2023-01-20T05:57:50.995235Z","iopub.execute_input":"2023-01-20T05:57:50.995818Z","iopub.status.idle":"2023-01-20T05:57:51.004177Z","shell.execute_reply.started":"2023-01-20T05:57:50.995778Z","shell.execute_reply":"2023-01-20T05:57:51.002986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"lower the learning rate and more epochs","metadata":{}},{"cell_type":"code","source":"\nnum_of_vecs = 64\nmin_count_events_per_session = 1\nnum_of_epochs = 8\ninit_lr = 0.01","metadata":{"execution":{"iopub.status.busy":"2023-01-20T05:57:54.795587Z","iopub.execute_input":"2023-01-20T05:57:54.796312Z","iopub.status.idle":"2023-01-20T05:57:54.801070Z","shell.execute_reply.started":"2023-01-20T05:57:54.796271Z","shell.execute_reply":"2023-01-20T05:57:54.800194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n#w2vec = Word2Vec(sentences=sentences, vector_size= 64, window = 3, negative = 8, ns_exponent = 0.2, sg = 1, min_count=1, workers=4)\nw2vec = Word2Vec(sentences, vector_size=num_of_vecs, min_count=min_count_events_per_session, workers=4, \n                 sg = 1, negative = 8, ns_exponent = 0.2, window=3, \n                 alpha = init_lr, epochs = num_of_epochs,\n                 seed = 3, hashfxn = hashf,\n                 compute_loss=True, callbacks=[w2vcallback()])\n    ","metadata":{"execution":{"iopub.status.busy":"2023-01-20T05:57:59.189896Z","iopub.execute_input":"2023-01-20T05:57:59.190395Z","iopub.status.idle":"2023-01-20T05:58:04.741229Z","shell.execute_reply.started":"2023-01-20T05:57:59.190357Z","shell.execute_reply":"2023-01-20T05:58:04.740033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nfrom annoy import AnnoyIndex\n\naid2idx = {aid: i for i, aid in enumerate(w2vec.wv.index_to_key)}\nindex = AnnoyIndex(64, 'euclidean')\n\nfor aid, idx in aid2idx.items():\n    index.add_item(idx, w2vec.wv.vectors[idx])\n    \nindex.build(32)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nfrom collections import defaultdict\nimport collections\n\nsession_types = ['clicks', 'carts', 'orders']\ntest_session_AIDs = test.to_pandas().reset_index(drop=True).groupby('session')['aid'].apply(list)\ntest_session_types = test.to_pandas().reset_index(drop=True).groupby('session')['type'].apply(list)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = []\n\ntype_weight_multipliers = {0: 1, 1: 6, 2: 3}\n\nsession_num = len(test_session_AIDs)\n\nfor AIDs, types in zip(test_session_AIDs[:session_num], test_session_types[:session_num]):\n    if len(AIDs) >= 20:\n        # if we have enough aids (over equals 20) we don't need to look for candidates! we just use the old logic\n        weights=np.logspace(0.1,1,len(AIDs),base=2, endpoint=True)-1\n        aids_temp=defaultdict(lambda: 0)\n        for aid,w,t in zip(AIDs,weights,types): \n            aids_temp[aid]+= w * type_weight_multipliers[t]\n            \n        sorted_aids=[k for k, v in sorted(aids_temp.items(), key=lambda item: -item[1])]\n        labels.append(sorted_aids[:20])\n    else:\n        # here we don't have 20 aids to output -- we will use word2vec embeddings to generate candidates!\n        AIDs = list(dict.fromkeys(AIDs[::-1]))\n        \n        # let's grab the up to 3 recent aids\n        recent_len = max(min(3,len(AIDs)),1)\n        \n        # how many aids for each aid\n        AIDs_num = round((20-len(AIDs))/recent_len) + 2\n        \n        # let's look for some neighbors!        \n        nns_it = []\n        for it in range(0,recent_len):\n            nns_it += [w2vec.wv.index_to_key[i] for i in index.get_nns_by_item(aid2idx[AIDs[it]], AIDs_num)[1:]]\n        \n        # select repeating and unique neighbors\n        nns_repeated = [item for item, count in collections.Counter(nns_it).items() if count > 1]\n        nns_once = [item for item, count in collections.Counter(nns_it).items() if count == 1]\n\n        # prepare selection\n        nns = (nns_repeated+nns_once)[:20]\n        labels.append((AIDs+nns)[:20])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_as_strings = [' '.join([str(l) for l in lls]) for lls in labels]\n\npredictions = pd.DataFrame(data={'session_type': test_session_AIDs.index, 'labels': labels_as_strings})\n\nprediction_dfs = []\n\nfor st in session_types:\n    modified_predictions = predictions.copy()\n    modified_predictions.session_type = modified_predictions.session_type.astype('str') + f'_{st}'\n    prediction_dfs.append(modified_predictions)\n\nsubmission = pd.concat(prediction_dfs).reset_index(drop=True)\nsubmission.to_csv('submission.csv', index=False)\n\ndel labels, labels_as_strings, predictions, prediction_dfs\ngc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# note\nother sections of code are based on @Word2vec model [training and submission | 0.533]\n","metadata":{}}]}