{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook builds on my previous work -- [co-visitation matrix - simplified, imprvd logic 🔥](https://www.kaggle.com/code/radek1/co-visitation-matrix-simplified-imprvd-logic?scriptVersionId=110068977) that achieves 0.558 on the LB.\n\nHere we take the functionality from that notebook, run on 1/1000 of the data (it achieves ~0.487 on public LB).\n\nThe next step in improving our results is to create a robust local validation framework to facilitate experimentation. This can be a stepping stone towards a much stronger result.\n\nLet's take a stab at implementing a local validation framework in this notebook!\n\n## Other resources you might find useful:\n\n* [💡 [2 methods] How-to ensemble predictions 🏅🏅🏅](https://www.kaggle.com/code/radek1/2-methods-how-to-ensemble-predictions)\n* [co-visitation matrix - simplified, imprvd logic 🔥](https://www.kaggle.com/code/radek1/co-visitation-matrix-simplified-imprvd-logic)\n* [💡 Word2Vec How-to [training and submission]🚀🚀🚀](https://www.kaggle.com/code/radek1/word2vec-how-to-training-and-submission)\n* [local validation tracks public LB perfecty -- here is the setup](https://www.kaggle.com/competitions/otto-recommender-system/discussion/364991)\n* [💡 For my friends from Twitter and LinkedIn -- here is how to dive into this competition 🐳](https://www.kaggle.com/competitions/otto-recommender-system/discussion/368560)\n* [Full dataset processed to CSV/parquet files with optimized memory footprint](https://www.kaggle.com/competitions/otto-recommender-system/discussion/363843)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\ntrain = pd.read_parquet('../input/otto-full-optimized-memory-footprint/train.parquet')\ntest = pd.read_parquet('../input/otto-full-optimized-memory-footprint/test.parquet')\n\n!pip install pickle5\nimport pickle5 as pickle\n\nwith open('../input/otto-full-optimized-memory-footprint/id2type.pkl', \"rb\") as fh:\n    id2type = pickle.load(fh)\nwith open('../input/otto-full-optimized-memory-footprint/type2id.pkl', \"rb\") as fh:\n    type2id = pickle.load(fh)\n    \nsample_sub = pd.read_csv('../input/otto-recommender-system/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-11-26T11:46:41.693406Z","iopub.execute_input":"2022-11-26T11:46:41.693756Z","iopub.status.idle":"2022-11-26T11:47:03.800118Z","shell.execute_reply.started":"2022-11-26T11:46:41.693684Z","shell.execute_reply":"2022-11-26T11:47:03.798890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DO_LOCAL_VALIDATION = True","metadata":{"execution":{"iopub.status.busy":"2022-11-26T11:47:03.801914Z","iopub.execute_input":"2022-11-26T11:47:03.802438Z","iopub.status.idle":"2022-11-26T11:47:03.808959Z","shell.execute_reply.started":"2022-11-26T11:47:03.802389Z","shell.execute_reply":"2022-11-26T11:47:03.807856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Local CV","metadata":{}},{"cell_type":"markdown","source":"For local CV, we will use the last weeks data of train for validation.\n\nEssentially, without modifying the calculations in the notebook, we can run evaluation locally if we replace the contents of the `train` and `test` variables.\n\nWhen doing local validation, we will print out local results. And without it, we will train on full data and submit to Kaggle LB.","metadata":{}},{"cell_type":"code","source":"%%time\n\nif DO_LOCAL_VALIDATION:\n    seven_days = 7*24*60*60\n\n    train_cutoff = train.ts.max() - seven_days\n    test = train[train.ts > train_cutoff]\n    train = train[train.ts <= train_cutoff]\n\n    # Let's discard overlapping sessions.\n    overlapping_sessions = set(train.session).intersection(set(test.session))\n\n    test = test[~test.session.isin(overlapping_sessions)]\n\n    new_test = []\n    data_to_calculate_validation_score = []\n\n    for grp in test.groupby('session'):\n        cutoff = np.random.randint(1, grp[1].shape[0]) # we want at least a single item in our validation data for each session\n        new_test.append(grp[1].iloc[:cutoff])\n        data_to_calculate_validation_score.append(grp[1].iloc[cutoff:])\n\n    test = pd.concat(new_test).reset_index(drop=True)\n    valid = pd.concat(data_to_calculate_validation_score).reset_index(drop=True)\n\n    test.to_parquet('_test.parquet')\n    valid.to_parquet('_valid.parquet')\n\n    del new_test, data_to_calculate_validation_score","metadata":{"execution":{"iopub.status.busy":"2022-11-26T11:47:03.811063Z","iopub.execute_input":"2022-11-26T11:47:03.811954Z","iopub.status.idle":"2022-11-26T11:56:52.149769Z","shell.execute_reply.started":"2022-11-26T11:47:03.811911Z","shell.execute_reply":"2022-11-26T11:56:52.148126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have now swapped the train and test sets for the ones we conjured and can now proceed to train as we would normally.","metadata":{}},{"cell_type":"markdown","source":"# Train","metadata":{}},{"cell_type":"code","source":"fraction_of_sessions_to_use = 1/1000","metadata":{"execution":{"iopub.status.busy":"2022-11-26T11:56:52.152683Z","iopub.execute_input":"2022-11-26T11:56:52.153116Z","iopub.status.idle":"2022-11-26T11:56:52.158562Z","shell.execute_reply.started":"2022-11-26T11:56:52.153084Z","shell.execute_reply":"2022-11-26T11:56:52.157389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nlucky_sessions_train = train.drop_duplicates(['session']).sample(frac=fraction_of_sessions_to_use)['session']\nsubset_of_train = train[train.session.isin(lucky_sessions_train)]\n\nlucky_sessions_test = test.drop_duplicates(['session']).sample(frac=fraction_of_sessions_to_use)['session']\nsubset_of_test = test[test.session.isin(lucky_sessions_test)]","metadata":{"execution":{"iopub.status.busy":"2022-11-26T11:56:52.160305Z","iopub.execute_input":"2022-11-26T11:56:52.160768Z","iopub.status.idle":"2022-11-26T11:57:01.726573Z","shell.execute_reply.started":"2022-11-26T11:56:52.160723Z","shell.execute_reply":"2022-11-26T11:57:01.725453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subset_of_train.index = pd.MultiIndex.from_frame(subset_of_train[['session']])\nsubset_of_test.index = pd.MultiIndex.from_frame(subset_of_test[['session']])","metadata":{"execution":{"iopub.status.busy":"2022-11-26T11:57:01.728132Z","iopub.execute_input":"2022-11-26T11:57:01.728514Z","iopub.status.idle":"2022-11-26T11:57:01.742115Z","shell.execute_reply.started":"2022-11-26T11:57:01.728482Z","shell.execute_reply":"2022-11-26T11:57:01.741013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nall_consecutive_AIDs = []\nchunk_size = 60_000\n\nsessions = subset_of_train.session.unique()\nfor i in range(0, sessions.shape[0], chunk_size):\n    current_chunk = subset_of_train.loc[sessions[i]:sessions[min(sessions.shape[0]-1, i+chunk_size-1)]].reset_index(drop=True)\n    current_chunk = current_chunk.groupby('session').apply(lambda g: g.tail(30)).reset_index(drop=True)\n    consecutive_AIDs = current_chunk.merge(current_chunk, on='session')\n    consecutive_AIDs = consecutive_AIDs[consecutive_AIDs.aid_x != consecutive_AIDs.aid_y]\n    consecutive_AIDs['days_elapsed'] = (consecutive_AIDs.ts_y - consecutive_AIDs.ts_x) / (24 * 60 * 60)\n    consecutive_AIDs = consecutive_AIDs[(consecutive_AIDs.days_elapsed > 0) & (consecutive_AIDs.days_elapsed <= 1)]\n    all_consecutive_AIDs.append(\n        consecutive_AIDs\n    )\n    \nsessions = subset_of_test.session.unique()\nfor i in range(0, sessions.shape[0], chunk_size):\n    current_chunk = subset_of_test.loc[sessions[i]:sessions[min(sessions.shape[0]-1, i+chunk_size-1)]].reset_index(drop=True)\n    current_chunk = current_chunk.groupby('session').apply(lambda g: g.tail(30)).reset_index(drop=True)\n    consecutive_AIDs = current_chunk.merge(current_chunk, on='session')\n    consecutive_AIDs = consecutive_AIDs[consecutive_AIDs.aid_x != consecutive_AIDs.aid_y]\n    consecutive_AIDs['days_elapsed'] = (consecutive_AIDs.ts_y - consecutive_AIDs.ts_x) / (24 * 60 * 60)\n    consecutive_AIDs = consecutive_AIDs[(consecutive_AIDs.days_elapsed > 0) & (consecutive_AIDs.days_elapsed <= 1)]\n    all_consecutive_AIDs.append(\n        consecutive_AIDs\n    )\n\nall_consecutive_AIDs = pd.concat(all_consecutive_AIDs).drop_duplicates(['session', 'aid_x', 'aid_y'])[['aid_x', 'aid_y']]","metadata":{"execution":{"iopub.status.busy":"2022-11-26T11:57:01.743789Z","iopub.execute_input":"2022-11-26T11:57:01.744358Z","iopub.status.idle":"2022-11-26T11:57:06.006484Z","shell.execute_reply.started":"2022-11-26T11:57:01.744317Z","shell.execute_reply":"2022-11-26T11:57:06.005268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nfrom collections import defaultdict, Counter\n\nnext_AIDs = defaultdict(Counter)\n\nfor row in all_consecutive_AIDs.itertuples():\n    next_AIDs[row.aid_x][row.aid_y] += 1","metadata":{"execution":{"iopub.status.busy":"2022-11-26T11:57:06.008040Z","iopub.execute_input":"2022-11-26T11:57:06.008438Z","iopub.status.idle":"2022-11-26T11:57:06.617119Z","shell.execute_reply.started":"2022-11-26T11:57:06.008402Z","shell.execute_reply":"2022-11-26T11:57:06.615581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(next_AIDs)","metadata":{"execution":{"iopub.status.busy":"2022-11-26T11:57:06.619007Z","iopub.execute_input":"2022-11-26T11:57:06.619371Z","iopub.status.idle":"2022-11-26T11:57:06.629119Z","shell.execute_reply.started":"2022-11-26T11:57:06.619339Z","shell.execute_reply":"2022-11-26T11:57:06.628007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now let's generate the predictions.","metadata":{"execution":{"iopub.status.busy":"2022-11-04T00:06:55.37083Z","iopub.execute_input":"2022-11-04T00:06:55.371203Z","iopub.status.idle":"2022-11-04T00:06:55.389977Z","shell.execute_reply.started":"2022-11-04T00:06:55.371178Z","shell.execute_reply":"2022-11-04T00:06:55.388167Z"}}},{"cell_type":"code","source":"%%time\n\ntest_session_AIDs = test.groupby('session')['aid'].apply(list)\nsession_types = ['clicks', 'carts', 'orders']\n\ntest_session_AIDs.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-26T11:57:06.631007Z","iopub.execute_input":"2022-11-26T11:57:06.631397Z","iopub.status.idle":"2022-11-26T11:57:39.874385Z","shell.execute_reply.started":"2022-11-26T11:57:06.631366Z","shell.execute_reply":"2022-11-26T11:57:39.873262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nlabels = []\n\nfor AIDs in test_session_AIDs:\n    AIDs = list(dict.fromkeys(AIDs[::-1]))\n    if len(AIDs) >= 20:\n        labels.append(AIDs[:20])\n    else:\n        counter = Counter()\n\n        for AID in AIDs:\n            subsequent_AID_counter = next_AIDs.get(AID)\n            if subsequent_AID_counter:\n                counter += subsequent_AID_counter\n        \n        AIDs += [AID for AID, cnt in counter.most_common(40) if AID not in AIDs]\n        labels.append(AIDs[:20])","metadata":{"execution":{"iopub.status.busy":"2022-11-26T11:57:39.876244Z","iopub.execute_input":"2022-11-26T11:57:39.877192Z","iopub.status.idle":"2022-11-26T11:58:25.233485Z","shell.execute_reply.started":"2022-11-26T11:57:39.877144Z","shell.execute_reply":"2022-11-26T11:58:25.232241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from matplotlib import pyplot as plt\n\nplt.hist([len(l) for l in labels]);","metadata":{"execution":{"iopub.status.busy":"2022-11-26T11:58:25.235021Z","iopub.execute_input":"2022-11-26T11:58:25.235421Z","iopub.status.idle":"2022-11-26T11:58:31.587529Z","shell.execute_reply.started":"2022-11-26T11:58:25.235385Z","shell.execute_reply":"2022-11-26T11:58:31.586312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_as_strings = [' '.join([str(l) for l in lls]) for lls in labels]\n\npredictions = pd.DataFrame(data={'session_type': test_session_AIDs.index, 'labels': labels_as_strings})","metadata":{"execution":{"iopub.status.busy":"2022-11-26T11:58:31.590630Z","iopub.execute_input":"2022-11-26T11:58:31.590990Z","iopub.status.idle":"2022-11-26T11:58:37.050734Z","shell.execute_reply.started":"2022-11-26T11:58:31.590959Z","shell.execute_reply":"2022-11-26T11:58:37.049640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_dfs = []\n\nfor st in session_types:\n    modified_predictions = predictions.copy()\n    modified_predictions.session_type = modified_predictions.session_type.astype('str') + f'_{st}'\n    prediction_dfs.append(modified_predictions)\n\nsubmission = pd.concat(prediction_dfs).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-11-26T11:58:37.052076Z","iopub.execute_input":"2022-11-26T11:58:37.052440Z","iopub.status.idle":"2022-11-26T11:58:41.631400Z","shell.execute_reply.started":"2022-11-26T11:58:37.052408Z","shell.execute_reply":"2022-11-26T11:58:41.630118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# We either submit to Kaggle or run validation locally\n\nWe need to now reverse the processing we applied to our predictions to shape them into a submission.\n\nI am undoing this work here on purpose. I will replace the code I use for predictions down the road, so I want my evaluation framework to work with data formatted like for making a submission.","metadata":{}},{"cell_type":"code","source":"if DO_LOCAL_VALIDATION:\n    submission['session'] = submission.session_type.apply(lambda x: int(x.split('_')[0]))\n    submission['type'] = submission.session_type.apply(lambda x: x.split('_')[1])\n    submission.labels = submission.labels.apply(lambda x: [int(i) for i in x.split(' ')][:20])\n\n    valid.type = valid.type.map(lambda idx: id2type[idx])\n    ground_truth = valid.groupby(['session', 'type'])['aid'].apply(list)\n    ground_truth = ground_truth.reset_index().rename(columns={'aid': 'labels'})\n    ground_truth.loc[ground_truth.type == 'clicks', 'labels'] = ground_truth.loc[ground_truth.type == 'clicks', 'labels'].str[:1]\n\n    submission_with_gt = submission.merge(ground_truth[['session', 'type', 'labels']], how='left', on=['session', 'type'])\n    submission_with_gt = submission_with_gt[~submission_with_gt.labels_y.isna()]\n    submission_with_gt['hits'] = submission_with_gt.apply(lambda df: len(set(df.labels_x).intersection(set(df.labels_y))), axis=1)\n    submission_with_gt['gt_count'] = submission_with_gt.labels_y.str.len().clip(0,20)\n\n    recall_per_type = submission_with_gt.groupby(['type'])['hits'].sum() / submission_with_gt.groupby(['type'])['gt_count'].sum() \n    local_validation_score = (recall_per_type * pd.Series({'clicks': 0.10, 'carts': 0.30, 'orders': 0.60})).sum()\n    print(f'Local validation score: {local_validation_score}')\n\nelse:\n    submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-11-26T11:58:41.633002Z","iopub.execute_input":"2022-11-26T11:58:41.633396Z","iopub.status.idle":"2022-11-26T12:01:12.034368Z","shell.execute_reply.started":"2022-11-26T11:58:41.633360Z","shell.execute_reply":"2022-11-26T12:01:12.033034Z"},"trusted":true},"execution_count":null,"outputs":[]}]}