{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\nimport cudf\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport scipy\nfrom scipy import stats\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport cupy\nimport glob\nimport time\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-23T19:50:00.368355Z","iopub.execute_input":"2022-12-23T19:50:00.369282Z","iopub.status.idle":"2022-12-23T19:50:01.648705Z","shell.execute_reply.started":"2022-12-23T19:50:00.369195Z","shell.execute_reply":"2022-12-23T19:50:01.647657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading and Analyzing JSON Data\n\nWe will begin by loading json data with a small subset of around 100k as the original dataset is quite big, and we may run out of RAM computing that.","metadata":{}},{"cell_type":"code","source":"%%time\ntrain_df = pd.DataFrame()\nchunks = pd.read_json('/kaggle/input/otto-recommender-system/train.jsonl', lines=True, chunksize=100_000)\n\nfor chunk in chunks:\n    event_dict = {'session': [], 'aid': [], 'ts': [], 'type': []}\n    \n    for session, events in zip(chunk['session'].tolist(), chunk['events'].tolist()):\n        for event in events:\n            event_dict['session'].append(session)\n            event_dict['aid'].append(event['aid'])\n            event_dict['ts'].append(event['ts'])\n            event_dict['type'].append(event['type'])\n    train_df = pd.DataFrame(event_dict)\n    \n    break\n        \ntrain_df = train_df.reset_index(drop=True)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-12-23T19:50:07.252215Z","iopub.execute_input":"2022-12-23T19:50:07.252665Z","iopub.status.idle":"2022-12-23T19:50:25.499214Z","shell.execute_reply.started":"2022-12-23T19:50:07.252628Z","shell.execute_reply":"2022-12-23T19:50:25.498018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**We observe that there are three types of events here clicks, carts and orders, so let's check if there are any correlation amongst these events, since we are looking at an Multi-Objective Recommender System, we would want to be able to estabilish how the user's purchasing history was determined for generating better recommendations.**\n\n**Since the features in consideration are categorical variables, we will use 2x2 contingency table and `corr()` metrics to estabilish correlation.**","metadata":{}},{"cell_type":"code","source":"ct = pd.crosstab(index = train_df['session'],columns=train_df['type'])\n\nsns.heatmap(ct.corr(), cmap='viridis', annot=True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-23T19:50:25.501628Z","iopub.execute_input":"2022-12-23T19:50:25.502026Z","iopub.status.idle":"2022-12-23T19:50:28.431702Z","shell.execute_reply.started":"2022-12-23T19:50:25.501987Z","shell.execute_reply":"2022-12-23T19:50:28.430807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**we observe that there is 61% probability of clicks being put into carts. 64% probability that the items in carts are being ordered.** ","metadata":{}},{"cell_type":"code","source":"del train_df, ct ","metadata":{"execution":{"iopub.status.busy":"2022-12-23T19:50:28.433018Z","iopub.execute_input":"2022-12-23T19:50:28.433528Z","iopub.status.idle":"2022-12-23T19:50:28.438816Z","shell.execute_reply.started":"2022-12-23T19:50:28.433487Z","shell.execute_reply":"2022-12-23T19:50:28.437724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading Data and Creating a Recommender System based on Matrix Factorization\n\n**We will proceed with using the otto-full-optimized-memory-footprint dataset since it is a compressed version with libraries like Merlin, and cuDF, we will proceed with creating a model**.","metadata":{}},{"cell_type":"code","source":"#we need to create columns specifically for type, time-stamps and aid which is the unique id\n# lets time the process\n%time\n\ntrain_df = cudf.read_parquet('/kaggle/input/otto-full-optimized-memory-footprint/train.parquet')\ntest_df = cudf.read_parquet('/kaggle/input/otto-full-optimized-memory-footprint/test.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-12-23T19:50:28.442117Z","iopub.execute_input":"2022-12-23T19:50:28.442811Z","iopub.status.idle":"2022-12-23T19:50:51.214363Z","shell.execute_reply.started":"2022-12-23T19:50:28.442766Z","shell.execute_reply":"2022-12-23T19:50:51.213391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We need to create aid-aid pairs to train our matrix factorization model!\n\nLet's us grab the pairs both from the train and test set.","metadata":{}},{"cell_type":"code","source":"%time\ntrain_pairs = cudf.concat([train_df, test_df])[['session','aid']]\n\ndel train_df, test_df\n\ntrain_pairs['aid_next'] = train_pairs.groupby('session').aid.shift(-1)\ntrain_pairs = train_pairs[['aid', 'aid_next']].dropna().reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T19:50:51.219271Z","iopub.execute_input":"2022-12-23T19:50:51.221594Z","iopub.status.idle":"2022-12-23T19:50:53.024622Z","shell.execute_reply.started":"2022-12-23T19:50:51.221538Z","shell.execute_reply":"2022-12-23T19:50:53.023597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"cardinality means possible values the feature can assume, so here it is possible that 'aid' column can have many values.","metadata":{}},{"cell_type":"code","source":"cardinality_aids = max(train_pairs['aid'].max(), train_pairs['aid_next'].max())\ncardinality_aids","metadata":{"execution":{"iopub.status.busy":"2022-12-23T19:50:53.026307Z","iopub.execute_input":"2022-12-23T19:50:53.026816Z","iopub.status.idle":"2022-12-23T19:50:53.060605Z","shell.execute_reply.started":"2022-12-23T19:50:53.026773Z","shell.execute_reply":"2022-12-23T19:50:53.059108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install merlin-dataloader==0.0.2","metadata":{"execution":{"iopub.status.busy":"2022-12-23T19:50:53.062156Z","iopub.execute_input":"2022-12-23T19:50:53.062617Z","iopub.status.idle":"2022-12-23T19:52:22.802068Z","shell.execute_reply.started":"2022-12-23T19:50:53.062575Z","shell.execute_reply":"2022-12-23T19:52:22.800782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from merlin.loader.torch import Loader \n\ntrain_pairs[:-10_000_000].to_pandas().to_parquet('train_pairs.parquet')\ntrain_pairs[-10_000_000:].to_pandas().to_parquet('valid_pairs.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-12-23T19:52:22.804444Z","iopub.execute_input":"2022-12-23T19:52:22.804913Z","iopub.status.idle":"2022-12-23T19:52:43.969200Z","shell.execute_reply.started":"2022-12-23T19:52:22.804862Z","shell.execute_reply":"2022-12-23T19:52:43.968012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from merlin.loader.torch import Loader \nfrom merlin.io import Dataset\n\ntrain_ds = Dataset('train_pairs.parquet')\ntrain_dl_merlin = Loader(train_ds, 65536, True)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T19:52:43.973459Z","iopub.execute_input":"2022-12-23T19:52:43.974280Z","iopub.status.idle":"2022-12-23T19:52:45.092554Z","shell.execute_reply.started":"2022-12-23T19:52:43.974240Z","shell.execute_reply":"2022-12-23T19:52:45.091290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%time\n\nfor batch in train_dl_merlin:\n    aid1, aid2 = batch[0], batch[1]","metadata":{"execution":{"iopub.status.busy":"2022-12-23T19:52:45.100350Z","iopub.execute_input":"2022-12-23T19:52:45.102948Z","iopub.status.idle":"2022-12-23T19:52:49.037083Z","shell.execute_reply.started":"2022-12-23T19:52:45.102900Z","shell.execute_reply":"2022-12-23T19:52:49.036094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Matrix Factorization\n\n**referencing the implementations of https://www.kaggle.com/code/cpmpml/matrix-factorization-with-gpu**","metadata":{}},{"cell_type":"code","source":"import torch\nfrom torch import nn\n\nclass MatrixFactorization(nn.Module):\n    def __init__(self, n_aids, n_factors):\n        super().__init__()\n        self.aid_factors = nn.Embedding(n_aids, n_factors, sparse=True)\n        \n    def forward(self, aid1, aid2):\n        aid1 = self.aid_factors(aid1)\n        aid2 = self.aid_factors(aid2)\n        \n        return (aid1 * aid2).sum(dim=1)\n    \nclass AverageMeter(object):\n    \"\"\"Computes and stores the average and current value\"\"\"\n    def __init__(self, name, fmt=':f'):\n        self.name = name\n        self.fmt = fmt\n        self.reset()\n\n    def reset(self):\n        self.val = 0\n        self.avg = 0\n        self.sum = 0\n        self.count = 0\n\n    def update(self, val, n=1):\n        self.val = val\n        self.sum += val * n\n        self.count += n\n        self.avg = self.sum / self.count\n\n    def __str__(self):\n        fmtstr = '{name} {val' + self.fmt + '} ({avg' + self.fmt + '})'\n        return fmtstr.format(**self.__dict__)\n\nvalid_ds = Dataset('valid_pairs.parquet')\nvalid_dl_merlin = Loader(valid_ds, 65536, True)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T19:52:49.038519Z","iopub.execute_input":"2022-12-23T19:52:49.038898Z","iopub.status.idle":"2022-12-23T19:52:49.273844Z","shell.execute_reply.started":"2022-12-23T19:52:49.038855Z","shell.execute_reply":"2022-12-23T19:52:49.272771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.optim import SparseAdam, AdamW\n\nnum_epochs=25\nlr=0.01\n\nmodel = MatrixFactorization(cardinality_aids+1, 64)\noptimizer = SparseAdam(model.parameters(), lr=lr)\ncriterion = nn.BCEWithLogitsLoss()","metadata":{"execution":{"iopub.status.busy":"2022-12-23T19:55:38.644508Z","iopub.execute_input":"2022-12-23T19:55:38.644894Z","iopub.status.idle":"2022-12-23T19:55:39.833798Z","shell.execute_reply.started":"2022-12-23T19:55:38.644860Z","shell.execute_reply":"2022-12-23T19:55:39.832621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmodel.to('cuda')\nfor epoch in range(num_epochs):\n    for batch, _ in train_dl_merlin:\n        model.train()\n        losses = AverageMeter('Loss', ':.4e')\n            \n        aid1, aid2 = batch['aid'], batch['aid_next']\n        aid1 = aid1.to('cuda')\n        aid2 = aid2.to('cuda')\n        output_pos = model(aid1, aid2)\n        output_neg = model(aid1, aid2[torch.randperm(aid2.shape[0])])\n        \n        output = torch.cat([output_pos, output_neg])\n        targets = torch.cat([torch.ones_like(output_pos), torch.zeros_like(output_pos)])\n        loss = criterion(output, targets)\n        losses.update(loss.item())\n        \n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n        \n    model.eval()\n    \n    with torch.no_grad():\n        accuracy = AverageMeter('accuracy')\n        for batch, _ in valid_dl_merlin:\n            aid1, aid2 = batch['aid'], batch['aid_next']\n            output_pos = model(aid1, aid2)\n            output_neg = model(aid1, aid2[torch.randperm(aid2.shape[0])])\n            accuracy_batch = torch.cat([output_pos.sigmoid() > 0.5, output_neg.sigmoid() < 0.5]).float().mean()\n            accuracy.update(accuracy_batch, aid1.shape[0])\n            \n    print(f'{epoch+1:02d}: * TrainLoss {losses.avg:.3f}  * Accuracy {accuracy.avg:.3f}')","metadata":{"execution":{"iopub.status.busy":"2022-12-23T19:55:40.889071Z","iopub.execute_input":"2022-12-23T19:55:40.889481Z","iopub.status.idle":"2022-12-23T20:10:05.506304Z","shell.execute_reply.started":"2022-12-23T19:55:40.889450Z","shell.execute_reply":"2022-12-23T20:10:05.504411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#getting the embeddings\n%time\nembeddings = model.aid_factors.weight.detach().cpu().numpy()\n\nfrom cuml.neighbors import NearestNeighbors\n\n\nknn = NearestNeighbors(n_neighbors=21, metric='euclidean')\nknn.fit(embeddings)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T20:10:05.508595Z","iopub.execute_input":"2022-12-23T20:10:05.509074Z","iopub.status.idle":"2022-12-23T20:10:06.055311Z","shell.execute_reply.started":"2022-12-23T20:10:05.509008Z","shell.execute_reply":"2022-12-23T20:10:06.054117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%time\n\n_, aid_nns = knn.kneighbors(embeddings)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T20:10:06.056986Z","iopub.execute_input":"2022-12-23T20:10:06.057394Z","iopub.status.idle":"2022-12-23T20:11:46.000847Z","shell.execute_reply.started":"2022-12-23T20:10:06.057356Z","shell.execute_reply":"2022-12-23T20:11:45.999808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import defaultdict\n\nsample_sub = pd.read_csv('../input/otto-recommender-system//sample_submission.csv')\ntest = cudf.read_parquet('../input/otto-full-optimized-memory-footprint/test.parquet')\n\nsession_types = ['clicks', 'carts', 'orders']\ngr = test.reset_index(drop=True).to_pandas().groupby('session')\ntest_session_AIDs = gr['aid'].apply(list)\ntest_session_types = gr['type'].apply(list)\n\nlabels = []\n\ntype_weight_multipliers = {0: 1, 1: 6, 2: 3}\nfor AIDs, types in zip(test_session_AIDs, test_session_types):\n    if len(AIDs) >= 20:\n        # if we have enough aids (over equals 20) we don't need to look for candidates! we just use the old logic\n        weights=np.logspace(0.1,1,len(AIDs),base=2, endpoint=True)-1\n        aids_temp=defaultdict(lambda: 0)\n        for aid,w,t in zip(AIDs,weights,types): \n            aids_temp[aid]+= w * type_weight_multipliers[t]\n            \n        sorted_aids=[k for k, v in sorted(aids_temp.items(), key=lambda item: -item[1])]\n        labels.append(sorted_aids[:20])\n    else:\n        # here we don't have 20 aids to output -- we will use approximate nearest neighbor search and our embeddings\n        # to generate candidates!\n        AIDs = list(dict.fromkeys(AIDs[::-1]))\n        \n        # let's grab the most recent aid\n        most_recent_aid = AIDs[0]\n        \n        # and look for some neighbors!\n        nns = list(aid_nns[most_recent_aid])\n                        \n        labels.append((AIDs+nns)[:20])","metadata":{"execution":{"iopub.status.busy":"2022-12-23T20:11:46.006420Z","iopub.execute_input":"2022-12-23T20:11:46.009299Z","iopub.status.idle":"2022-12-23T20:13:08.354745Z","shell.execute_reply.started":"2022-12-23T20:11:46.009253Z","shell.execute_reply":"2022-12-23T20:13:08.353337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_as_strings = [' '.join([str(l) for l in lls]) for lls in labels]\n\npredictions = pd.DataFrame(data={'session_type': test_session_AIDs.index, 'labels': labels_as_strings})\n\nprediction_dfs = []\n\nfor st in session_types:\n    modified_predictions = predictions.copy()\n    modified_predictions.session_type = modified_predictions.session_type.astype('str') + f'_{st}'\n    prediction_dfs.append(modified_predictions)\n\nsubmission = pd.concat(prediction_dfs).reset_index(drop=True)\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T20:13:08.356970Z","iopub.execute_input":"2022-12-23T20:13:08.357466Z","iopub.status.idle":"2022-12-23T20:13:48.994592Z","shell.execute_reply.started":"2022-12-23T20:13:08.357417Z","shell.execute_reply":"2022-12-23T20:13:48.991311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}