{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nimport torch\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport gc\ntorch.__version__","metadata":{"papermill":{"duration":0.018314,"end_time":"2022-11-17T15:15:23.794011","exception":false,"start_time":"2022-11-17T15:15:23.775697","status":"completed"},"tags":[],"id":"9d5fe4f7","outputId":"147cbc15-3c8c-4995-f4f2-ab3ba5094814","execution":{"iopub.status.busy":"2022-11-24T12:28:25.659099Z","iopub.execute_input":"2022-11-24T12:28:25.659564Z","iopub.status.idle":"2022-11-24T12:28:28.725258Z","shell.execute_reply.started":"2022-11-24T12:28:25.659525Z","shell.execute_reply":"2022-11-24T12:28:28.723833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This notebook presents a simple and fast solution for the recommendation task.\n\nI inspired my work from RFM Analysis ( Recency, Frequency, Money ) that is widely used in industry. However, in this problem we don't have money, so I replace it by the general popularity of each item/aid.\n\nI also noticed that the event type by itself can be used as a feature to build the overall score by which we're going to sort our recommended items. For exemple, aids putted into carts have higher probability to be ordered, than clicked aids, than purchased aids. This trick increased the overall score in the validation by 0.01.\n\nI also recommend to take a look at ","metadata":{}},{"cell_type":"code","source":"'''path = \"/kaggle/input/otto-full-optimized-memory-footprint/test.parquet\"\ntest_df = pd.read_parquet(path)\n\n\npath = \"/kaggle/input/otto-full-optimized-memory-footprint/train.parquet\"\ntrain_df = pd.read_parquet(path)\n\ntrain_aids_unique = pd.Series(train_df['aid'].unique())\ntest_aids_unique = pd.Series(test_df['aid'].unique())\ntest_aids_unique.isin(train_aids_unique).all()'''","metadata":{"execution":{"iopub.status.busy":"2022-11-24T12:28:28.728234Z","iopub.execute_input":"2022-11-24T12:28:28.729093Z","iopub.status.idle":"2022-11-24T12:28:28.738369Z","shell.execute_reply.started":"2022-11-24T12:28:28.729040Z","shell.execute_reply":"2022-11-24T12:28:28.736930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_X_path = \"/kaggle/input/otto-train-and-test-data-for-local-validation/test.parquet\"\ndf_ = pd.read_parquet(valid_X_path)","metadata":{"id":"ga6DC12bMto_","execution":{"iopub.status.busy":"2022-11-24T12:38:28.485702Z","iopub.execute_input":"2022-11-24T12:38:28.486164Z","iopub.status.idle":"2022-11-24T12:38:30.408037Z","shell.execute_reply.started":"2022-11-24T12:38:28.486130Z","shell.execute_reply":"2022-11-24T12:38:30.407064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Discretization of Recency / Frequency / General Popularity","metadata":{}},{"cell_type":"code","source":"# set the bins [<min, x, x, max]\ndef quantile_bins(df,col):\n    quantile_bins_ = [0, \n                    df[col].quantile(0.25),\n                    #df[col].quantile(0.5),\n                    df[col].quantile(0.75),\n                    df[col].quantile(0.9),\n                    df[col].max()]\n    return quantile_bins_","metadata":{"execution":{"iopub.status.busy":"2022-11-24T12:38:30.410265Z","iopub.execute_input":"2022-11-24T12:38:30.410686Z","iopub.status.idle":"2022-11-24T12:38:30.417365Z","shell.execute_reply.started":"2022-11-24T12:38:30.410651Z","shell.execute_reply":"2022-11-24T12:38:30.415690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Event processing","metadata":{}},{"cell_type":"markdown","source":"This preprocessing is super simple. I considered that there is more chance that an item will be ordered if it was put into the cart than clicked than bought.\n\nPriorities :\n\n- For orders -> Carts ( 2 ) / Click ( 1 ) / purchased ( 0 )\n\nThis trick helped me increase the public score by 0.01\n\nI tried to do same for `click` and `cart`, but it decreased the validation score.","metadata":{}},{"cell_type":"code","source":"def event_processing(df_):\n    dict_order = {0:1,\n         1:2,\n         2:0}\n\n    df_['event_order'] = df_['type'].map(dict_order)\n    \n    return df_","metadata":{"execution":{"iopub.status.busy":"2022-11-24T12:38:30.419191Z","iopub.execute_input":"2022-11-24T12:38:30.419878Z","iopub.status.idle":"2022-11-24T12:38:30.436669Z","shell.execute_reply.started":"2022-11-24T12:38:30.419822Z","shell.execute_reply":"2022-11-24T12:38:30.435083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## This function helps into creating the submission file\n\ndef click_cart_order(data):\n    data_list = []\n    for row in data.to_dict(orient=\"records\"):\n        session_id = row['session']\n        sorted_aids = row['sequence_labels_order']\n\n        data_list.append([f\"{session_id}_clicks\", sorted_aids])\n        data_list.append([f\"{session_id}_carts\", sorted_aids])\n        data_list.append([f\"{session_id}_orders\", sorted_aids])\n    sub = pd.DataFrame(\n    data_list, columns=[\"session_type\",'labels'])\n    return sub\n\n# Returned output example :\n# session_id_clicks [item1 item2 item3 ...]\n# session_id_carts  [item1 item2 item3 ...]\n# session_id_orders [item1 item2 item3 ...]","metadata":{"execution":{"iopub.status.busy":"2022-11-24T12:38:30.440087Z","iopub.execute_input":"2022-11-24T12:38:30.440806Z","iopub.status.idle":"2022-11-24T12:38:30.451433Z","shell.execute_reply.started":"2022-11-24T12:38:30.440766Z","shell.execute_reply":"2022-11-24T12:38:30.450514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def generate_prediction(df_,cond=\"validation\"):\n    product_ids =  df_.drop_duplicates(subset=['session','aid'])['aid'].tolist()\n    my_dict = pd.Series(product_ids).value_counts().to_dict()\n    ## general popularity of each aid  among all session\n    df_['general_popularity'] =  df_['aid'].map(my_dict)\n    ## Recency of aid per session\n    df_['recency'] = (df_.groupby(['session']).cumcount()+1).values\n    ## Frequency of aids per session\n    df_['freq_by_session'] = df_.groupby(['session', 'aid']).transform(\"count\").values[:,0]\n    \n    df_ = event_processing(df_)\n    \n    \n    ## Discretization of features\n    \n    df_['r_score'] = pd.cut(df_['recency'], quantile_bins(df_,\"recency\"), labels = [1, 2, 3,4])\n    df_['f_score'] = pd.cut(df_['freq_by_session'], quantile_bins(df_,\"freq_by_session\"), labels = [1, 2, 3,4])\n    df_['m_score'] = pd.cut(df_['general_popularity'], quantile_bins(df_,\"general_popularity\"), labels = [1, 2, 3,4])\n    \n    ## Creation of a global score\n    df_['rfm_group_order'] = ((df_['event_order'].astype(\"str\"))\n                   + (df_['f_score'].astype(\"str\"))\n                   + (df_['r_score'].astype(\"str\"))\n                   + df_['m_score'].astype(\"str\")).astype(\"int\")\n\n    \n    if cond == \"test\":\n    # For a matter of submission labels formats    \n        sort_score_order = (df_.sort_values(\"rfm_group_order\",ascending=False).groupby([\"session\"])['aid'].apply(list).apply(lambda x: str(list(dict.fromkeys(x)))))\n\n    else:\n        sort_score_order = (df_.sort_values(\"rfm_group_order\",ascending=False).groupby([\"session\"])['aid'].apply(list).apply(lambda x: list(dict.fromkeys(x))))\n\n\n    df = pd.DataFrame()\n    df['session'] = sort_score_order.index\n    df['sequence_labels_order'] = sort_score_order.values\n\n    sub = click_cart_order(df)\n\n    return sub\n","metadata":{"execution":{"iopub.status.busy":"2022-11-24T12:45:55.570304Z","iopub.execute_input":"2022-11-24T12:45:55.570757Z","iopub.status.idle":"2022-11-24T12:45:55.585274Z","shell.execute_reply.started":"2022-11-24T12:45:55.570720Z","shell.execute_reply":"2022-11-24T12:45:55.583746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = generate_prediction(df_,cond=\"validation\")","metadata":{"execution":{"iopub.status.busy":"2022-11-24T12:38:30.469568Z","iopub.execute_input":"2022-11-24T12:38:30.470890Z","iopub.status.idle":"2022-11-24T12:39:57.378169Z","shell.execute_reply.started":"2022-11-24T12:38:30.470835Z","shell.execute_reply":"2022-11-24T12:39:57.377193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-24T12:39:57.380279Z","iopub.execute_input":"2022-11-24T12:39:57.381222Z","iopub.status.idle":"2022-11-24T12:39:57.396155Z","shell.execute_reply.started":"2022-11-24T12:39:57.381182Z","shell.execute_reply":"2022-11-24T12:39:57.394444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Validation","metadata":{}},{"cell_type":"markdown","source":"This code was copied from the awesome discussion of @Radek [link](https://www.kaggle.com/competitions/otto-recommender-system/discussion/364991)  that helped me do validation of the method on the 7 last days of the train set","metadata":{}},{"cell_type":"code","source":"submission = pd.DataFrame()\nsubmission['session'] = sub.session_type.apply(lambda x: int(x.split('_')[0]))\nsubmission['type'] = sub.session_type.apply(lambda x: x.split('_')[1])\nsubmission['labels'] = sub.labels.apply(lambda x : [item for item in x[:20] ]) #.apply(lambda x: [int(i) for i in x.split(',')[:20]])\ntest_labels = pd.read_parquet('/kaggle/input/otto-train-and-test-data-for-local-validation/test_labels.parquet')\ntest_labels = test_labels.merge(submission, how='left', on=['session', 'type'])\ntest_labels['hits'] = test_labels.apply(lambda df: len(set(df.ground_truth).intersection(set(df.labels))), axis=1)\ntest_labels['gt_count'] = test_labels.ground_truth.str.len().clip(0,20)\nrecall_per_type = test_labels.groupby(['type'])['hits'].sum() / test_labels.groupby(['type'])['gt_count'].sum() \n","metadata":{"execution":{"iopub.status.busy":"2022-11-24T12:39:57.398248Z","iopub.execute_input":"2022-11-24T12:39:57.398667Z","iopub.status.idle":"2022-11-24T12:41:17.095492Z","shell.execute_reply.started":"2022-11-24T12:39:57.398613Z","shell.execute_reply":"2022-11-24T12:41:17.094171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score = (recall_per_type * pd.Series({'clicks': 0.1, 'carts': 0.30, 'orders': 0.60})).sum()\nscore","metadata":{"execution":{"iopub.status.busy":"2022-11-24T12:41:17.097690Z","iopub.execute_input":"2022-11-24T12:41:17.098353Z","iopub.status.idle":"2022-11-24T12:41:17.109981Z","shell.execute_reply.started":"2022-11-24T12:41:17.098303Z","shell.execute_reply":"2022-11-24T12:41:17.108675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inference","metadata":{}},{"cell_type":"markdown","source":"We simply compute the same pipeline for the test set","metadata":{}},{"cell_type":"code","source":"test_path = \"/kaggle/input/otto-full-optimized-memory-footprint/test.parquet\"\ntest_df = pd.read_parquet(test_path)","metadata":{"execution":{"iopub.status.busy":"2022-11-24T12:44:07.109694Z","iopub.execute_input":"2022-11-24T12:44:07.110075Z","iopub.status.idle":"2022-11-24T12:44:07.316793Z","shell.execute_reply.started":"2022-11-24T12:44:07.110044Z","shell.execute_reply":"2022-11-24T12:44:07.315666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = generate_prediction(test_df,cond=\"test\")\nsub['labels'] = sub['labels'].str.replace(\"[\",\"\").str.replace(\"]\",\"\").str.replace(\",\",\"\")","metadata":{"execution":{"iopub.status.busy":"2022-11-24T12:45:58.726402Z","iopub.execute_input":"2022-11-24T12:45:58.727613Z","iopub.status.idle":"2022-11-24T12:47:30.136224Z","shell.execute_reply.started":"2022-11-24T12:45:58.727544Z","shell.execute_reply":"2022-11-24T12:47:30.134893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-11-24T12:28:57.582569Z","iopub.status.idle":"2022-11-24T12:28:57.583489Z","shell.execute_reply.started":"2022-11-24T12:28:57.583263Z","shell.execute_reply":"2022-11-24T12:28:57.583287Z"},"trusted":true},"execution_count":null,"outputs":[]}]}