{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# install cuml\n\nimport sys\n!cp ../input/rapids/rapids.21.06 /opt/conda/envs/rapids.tar.gz\n!cd /opt/conda/envs/ && tar -xzvf rapids.tar.gz > /dev/null\nsys.path = [\"/opt/conda/envs/rapids/lib/python3.7/site-packages\"] + sys.path\nsys.path = [\"/opt/conda/envs/rapids/lib/python3.7\"] + sys.path\nsys.path = [\"/opt/conda/envs/rapids/lib\"] + sys.path \n!cp /opt/conda/envs/rapids/lib/libxgboost.so /opt/conda/lib/","metadata":{"execution":{"iopub.status.busy":"2023-08-01T14:08:19.086823Z","iopub.execute_input":"2023-08-01T14:08:19.087627Z","iopub.status.idle":"2023-08-01T14:10:35.876098Z","shell.execute_reply.started":"2023-08-01T14:08:19.087596Z","shell.execute_reply":"2023-08-01T14:10:35.874679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get test data\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nfrom pathlib import Path\n\ndata_path = Path('/kaggle/input/recsys-dataset/')\n\ntest_sessions = pd.DataFrame()\nchunks = pd.read_json(data_path / 'otto-recsys-test.jsonl', lines=True, chunksize=100_000)\n\nfor e, chunk in enumerate(chunks):\n    event_dict = {\n        'session': [],\n        'aid': [],\n        'ts': [],\n        'type': [],\n    }\n    if e < 2:\n        for session, events in zip(chunk['session'].tolist(), chunk['events'].tolist()):\n            for event in events:\n                event_dict['session'].append(session)\n                event_dict['aid'].append(event['aid'])\n                event_dict['ts'].append(event['ts'])\n                event_dict['type'].append(event['type'])\n        chunk_session = pd.DataFrame(event_dict)\n        test_sessions = pd.concat([test_sessions, chunk_session])\n    else:\n        break\n        \n\ntest_sessions = pl.from_pandas(test_sessions.reset_index(drop=True))\ntest_sessions = test_sessions.groupby('session').agg(pl.all()).sort(by='session')\n\n# Split test data into testA (session up to certain point) and testB (prediction)\ndictA = {'session': [], 'aid': [], 'ts': [], 'type': []}\ndictB = {'session': [], 'aid': [], 'ts': [], 'type': []}\n\nfor row in test_sessions.iter_rows():\n    split_idx = np.random.randint(1,len(row[1]))\n    dictA['session'].append(row[0])\n    dictA['aid'].append(row[1][:split_idx])\n    dictA['ts'].append(row[2][:split_idx])\n    dictA['type'].append(row[3][:split_idx])\n\n    dictB['session'].append(row[0])\n    dictB['aid'].append(row[1][split_idx:])\n    dictB['ts'].append(row[2][split_idx:])\n    dictB['type'].append(row[3][split_idx:])\n    \ntestA = pl.DataFrame(data=dictA).head(100)\ntestB = pl.DataFrame(data=dictB)\n\nactual_events = testB.head(100).explode(['aid', 'ts', 'type'])","metadata":{"execution":{"iopub.status.busy":"2023-08-04T02:56:19.748761Z","iopub.execute_input":"2023-08-04T02:56:19.749220Z","iopub.status.idle":"2023-08-04T02:56:47.249787Z","shell.execute_reply.started":"2023-08-04T02:56:19.749166Z","shell.execute_reply":"2023-08-04T02:56:47.248644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain = pl.read_parquet('../input/otto-train-and-test-data-for-local-validation/train.parquet')\n\n# Get subset of data \nfraction_of_sessions = 1\n\ntrain_sessions = train['session'].sample(fraction=fraction_of_sessions, seed=42)\ntrain = train.filter(pl.col(\"session\").is_in(train_sessions))\ntrain = train.sort(\"session\")\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-04T02:57:23.580209Z","iopub.execute_input":"2023-08-04T02:57:23.581173Z","iopub.status.idle":"2023-08-04T02:57:55.282461Z","shell.execute_reply.started":"2023-08-04T02:57:23.581126Z","shell.execute_reply":"2023-08-04T02:57:55.281372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# decrease aid amount\n# simple method based on counts\ndef top_n_interacted_aids(df, n):\n    return df['aid'].value_counts().head(n)['aid'].to_list()\n    \ncandidates = top_n_interacted_aids(train, 500000)\ncandidate_train = train.filter(pl.col('aid').is_in(candidates))\n","metadata":{"_kg_hide-output":false,"execution":{"iopub.status.busy":"2023-08-04T02:58:02.656132Z","iopub.execute_input":"2023-08-04T02:58:02.656834Z","iopub.status.idle":"2023-08-04T02:58:19.809917Z","shell.execute_reply.started":"2023-08-04T02:58:02.656801Z","shell.execute_reply":"2023-08-04T02:58:19.808905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# interaction matrix\n\n# count click,cart,buy for each (session, aid) pair\ncounts = candidate_train.groupby(['session', 'aid', 'type']) \\\n       .agg(click_count = pl.col('type').filter(pl.col('type') == 0).count(),\n            cart_count = pl.col('type').filter(pl.col('type') == 1).count(),\n            buy_count = pl.col('type').filter(pl.col('type') == 2).count())\n\n# calculate a value based on counts\ncounts = counts.with_columns(value = 1 * pl.col('click_count') + 10 * pl.col('cart_count') + 50 * pl.col('buy_count'))\ncounts = counts.with_columns(counts['value'].cast(pl.datatypes.UInt16)) \\\n       .drop(['type', 'click_count', 'cart_count', 'buy_count'])\ncounts = counts.sort(by='value')\n\n# convert to sparse matrix\n\nfrom scipy.sparse import coo_matrix\n\nrow = counts.get_column('session').to_numpy()\ncol = counts.get_column('aid').to_numpy()\ndata = counts.get_column('value').to_numpy().astype(np.float32)\ninteraction_matrix = coo_matrix((data, (row, col)))\naid_dimension = interaction_matrix.shape[1]\n","metadata":{"execution":{"iopub.status.busy":"2023-08-04T02:58:23.264135Z","iopub.execute_input":"2023-08-04T02:58:23.264629Z","iopub.status.idle":"2023-08-04T02:58:46.554773Z","shell.execute_reply.started":"2023-08-04T02:58:23.264589Z","shell.execute_reply":"2023-08-04T02:58:46.553743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cuml\nfrom sklearn.neighbors import NearestNeighbors\n\nmodel = NearestNeighbors(n_neighbors=50000, metric='manhattan')\ninteraction_matrix = interaction_matrix.tocsr()\nmodel.fit(interaction_matrix)\n\ndef predict_20(user, csr_mat, model):\n    # get n nearest users\n    distances, neighbour_indices = model.kneighbors(user)\n\n    # get items from neighbours that user has not interacted with\n    neighbour_aids = dict()\n    for idx in neighbour_indices[0]:\n        row = csr_mat.getrow(idx)\n\n        for i, idx in enumerate(row.indices):\n            if idx in neighbour_aids:\n                neighbour_aids[idx] += row.data[i]\n            else:\n                neighbour_aids[idx] = row.data[i]\n    \n    top_aids = sorted(neighbour_aids, key=lambda x: neighbour_aids[x], reverse=True)\n    return top_aids[:20]\n","metadata":{"execution":{"iopub.status.busy":"2023-08-04T03:24:35.479088Z","iopub.execute_input":"2023-08-04T03:24:35.480101Z","iopub.status.idle":"2023-08-04T03:24:35.662865Z","shell.execute_reply.started":"2023-08-04T03:24:35.480069Z","shell.execute_reply":"2023-08-04T03:24:35.661844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# hyperparameter tuning\n\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.model_selection import cross_val_score\n\ndef average_distance_metric(estimator, X):\n    # Fit the estimator on the data and get the distances and indices of the nearest neighbors\n    estimator.fit(X)\n    distances, _ = estimator.kneighbors(X)\n    \n    # Calculate the average distance of the nearest neighbors\n    avg_distance = np.mean(np.mean(distances, axis=1))\n    return avg_distance\n\nn_neighbors = [10, 100, 1000, 5000, 10000, 50000]\nmetrics = ['cosine', 'manhattan', 'euclidean']\nbest_score = np.inf\nbest_n_neighbors = None\nbest_metric = None\n\nk_folds = 5\nfor n in n_neighbors:\n    for metric in metrics:\n        nn_model = NearestNeighbors(n_neighbors=n, metric=metric)\n        nn_model.fit(interaction_matrix)\n        \n        scores = cross_val_score(model, interaction_matrix, cv=k_folds, scoring=‘neg_mean_squared_error’)\n        average_distance = np.mean(scores)\n     \n        distances, _ = nn_model.kneighbors(interaction_matrix[:100, :])\n        average_distance = np.mean(np.mean(distances, axis=1))\n        \n        print(average_distance)\n        \n        if average_distance < best_score:\n            best_score = average_distance\n            best_n_neighbors = n\n            best_metric = metric\n","metadata":{"execution":{"iopub.status.busy":"2023-08-04T04:03:31.413639Z","iopub.execute_input":"2023-08-04T04:03:31.414082Z","iopub.status.idle":"2023-08-04T04:16:02.510391Z","shell.execute_reply.started":"2023-08-04T04:03:31.414049Z","shell.execute_reply":"2023-08-04T04:16:02.509202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predicting test\n\naction_weights = {'clicks': 1, 'carts': 10, 'orders': 50}\n\nsessions = []\npred = []\ntypes = [['clicks', 'carts', 'orders'] for _ in range(100)]\n\n# predict for each user in test set\nfor user in testA.rows(named=True):\n    # calculate user - aid interaction value\n    values = dict()\n    for aid, action in zip(user['aid'], user['type']):\n        if aid in values:\n            values[aid] += action_weights[action]\n        else:\n            values[aid] = action_weights[action]\n            \n    col = []\n    data = []\n    for aid, value in values.items():\n        col.append(aid)\n        data.append(value)\n    row = [0 for _ in range(len(col))]\n    \n    coo_row = coo_matrix((data, (row, col)), shape=(1, aid_dimension))\n    recommendations = predict_20(coo_row, interaction_matrix, model)\n    \n    sessions.append(user['session'])\n    pred.append(recommendations)\n\n# prediction df for evaluation\npred = pl.DataFrame({'session': sessions, 'pred_labels': pred, 'type': types})\npred = pred.explode('type')\n","metadata":{"execution":{"iopub.status.busy":"2023-08-04T03:24:44.121400Z","iopub.execute_input":"2023-08-04T03:24:44.121801Z","iopub.status.idle":"2023-08-04T03:31:11.800346Z","shell.execute_reply.started":"2023-08-04T03:24:44.121767Z","shell.execute_reply":"2023-08-04T03:31:11.799233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# evaluate score\npreds = pred\n\nid2type = {0: 'clicks', 1: 'carts', 2: 'orders'}\ntype2id = {'clicks': 0, 'carts': 1, 'orders': 2}\n\ngt = actual_events.groupby(['session', 'type']).agg(pl.col('aid')).rename({'aid': 'gt_labels'}).sort(by='session').with_columns(pl.col('type'))\ngt = gt.to_pandas()\ngt.loc[gt.type == 'clicks', 'gt_labels'] = gt.loc[gt.type == 'clicks', 'gt_labels'].str[:1]\ngt = pl.from_pandas(gt)\n\npreds_and_gt = gt.join(preds, how='left', on=['session', 'type']).with_columns(\n    hits = pl.col('pred_labels').list.intersection('gt_labels').list.lengths(),\n    gt_count = pl.col('gt_labels').list.lengths()\n)\n\npreds_and_gt = preds_and_gt.to_pandas()\npreds_and_gt.loc[preds_and_gt.type == 'carts', 'gt_labels'] = preds_and_gt.loc[preds_and_gt.type == 'carts', 'gt_labels'].str[:20]\npreds_and_gt.loc[preds_and_gt.type == 'orders', 'gt_labels'] = preds_and_gt.loc[preds_and_gt.type == 'orders', 'gt_labels'].str[:20]\npreds_and_gt = pl.from_pandas(preds_and_gt)\npreds_and_gt = preds_and_gt.with_columns(\n    gt_count = pl.col('gt_labels').list.lengths()\n)\n\nrecall_per_type = preds_and_gt.groupby('type').agg(recall = pl.col('hits').sum() / pl.col('gt_count').sum())\nlocal_validation_score = 0\nweights = {'clicks': 0.1, 'carts': 0.3, 'orders': 0.6}\nfor row in recall_per_type.rows(named=True):\n    local_validation_score += row['recall'] * weights[row['type']]\n\nprint(f': {local_validation_score}')","metadata":{"execution":{"iopub.status.busy":"2023-08-04T03:44:22.059061Z","iopub.execute_input":"2023-08-04T03:44:22.059934Z","iopub.status.idle":"2023-08-04T03:44:22.100705Z","shell.execute_reply.started":"2023-08-04T03:44:22.059900Z","shell.execute_reply":"2023-08-04T03:44:22.099601Z"},"trusted":true},"execution_count":null,"outputs":[]}]}