{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Based on EDWARD CROOKENDEN's Notebook \n\nNotebooks link: https://www.kaggle.com/code/edwardcrookenden/otto-getting-started-eda-baseline\n\nThis one is not completed. Most of the works are for me to learn from start. ","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\nfrom pathlib import Path\nimport pandas as pd\nimport random\nimport numpy as np\nimport json\nfrom datetime import timedelta\nfrom collections import Counter\nfrom tqdm.notebook import tqdm\nfrom heapq import nlargest\n\nimport warnings\nwarnings.filterwarnings('ignore')\nfrom collections import Counter\nfrom random import sample\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-31T12:50:54.647071Z","iopub.execute_input":"2023-01-31T12:50:54.647544Z","iopub.status.idle":"2023-01-31T12:50:54.660185Z","shell.execute_reply.started":"2023-01-31T12:50:54.647507Z","shell.execute_reply":"2023-01-31T12:50:54.658937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Paths ###\n\nDATA_PATH = Path('../input/otto-recommender-system')\nTRAIN_PATH = DATA_PATH/'train.jsonl'\nTEST_PATH = DATA_PATH/'test.jsonl'\nSAMPLE_SUB_PATH = Path('../input/otto-recommender-system/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-01-31T12:50:58.572813Z","iopub.execute_input":"2023-01-31T12:50:58.573253Z","iopub.status.idle":"2023-01-31T12:50:58.579641Z","shell.execute_reply.started":"2023-01-31T12:50:58.573217Z","shell.execute_reply":"2023-01-31T12:50:58.578516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## EDA and baseline","metadata":{}},{"cell_type":"code","source":"# check on the train data\n# with open(TRAIN_PATH, 'r') as f:\n#     print(f\"We have {len(f.readlines()):,} lines in the training data\")\n    \n# Too many lines, Not gonna run it.\n# We have 12,899,779 lines in the training data","metadata":{"execution":{"iopub.status.busy":"2023-01-31T12:51:16.046650Z","iopub.execute_input":"2023-01-31T12:51:16.047133Z","iopub.status.idle":"2023-01-31T12:51:16.052241Z","shell.execute_reply.started":"2023-01-31T12:51:16.047094Z","shell.execute_reply":"2023-01-31T12:51:16.051336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# read jsonl\nsample_size = 1000\n\nchunks = pd.read_json(TRAIN_PATH, lines=True, chunksize = sample_size)\n\nfor c in chunks:\n    sample_train_df = c\n    break\n\n#session id starts from 0","metadata":{"execution":{"iopub.status.busy":"2023-01-31T12:51:21.453499Z","iopub.execute_input":"2023-01-31T12:51:21.453946Z","iopub.status.idle":"2023-01-31T12:51:21.572888Z","shell.execute_reply.started":"2023-01-31T12:51:21.453896Z","shell.execute_reply":"2023-01-31T12:51:21.571865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Rename the id column with session number, session id is useful and can be changed\nsample_train_df.set_index('session', drop=True, inplace=True)\nsample_train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-31T12:51:24.004872Z","iopub.execute_input":"2023-01-31T12:51:24.005728Z","iopub.status.idle":"2023-01-31T12:51:24.103277Z","shell.execute_reply.started":"2023-01-31T12:51:24.005679Z","shell.execute_reply":"2023-01-31T12:51:24.101941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#events is the only column in the df\nsample_train_df.columns","metadata":{"execution":{"iopub.status.busy":"2023-01-31T12:51:26.720507Z","iopub.execute_input":"2023-01-31T12:51:26.720928Z","iopub.status.idle":"2023-01-31T12:51:26.727311Z","shell.execute_reply.started":"2023-01-31T12:51:26.720876Z","shell.execute_reply":"2023-01-31T12:51:26.726496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Events IS A LIST\n# Inside Events is a dict\nlen(sample_train_df['events'][0])\ntype(sample_train_df['events'][0][0])","metadata":{"execution":{"iopub.status.busy":"2023-01-31T12:51:29.412752Z","iopub.execute_input":"2023-01-31T12:51:29.413198Z","iopub.status.idle":"2023-01-31T12:51:29.422133Z","shell.execute_reply.started":"2023-01-31T12:51:29.413161Z","shell.execute_reply":"2023-01-31T12:51:29.420811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"session0 = sample_train_df['events'][0]\n#276 items","metadata":{"execution":{"iopub.status.busy":"2023-01-31T12:51:32.683005Z","iopub.execute_input":"2023-01-31T12:51:32.683464Z","iopub.status.idle":"2023-01-31T12:51:32.688627Z","shell.execute_reply.started":"2023-01-31T12:51:32.683418Z","shell.execute_reply":"2023-01-31T12:51:32.687473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sample_train_df['events'][0]","metadata":{"execution":{"iopub.status.busy":"2023-01-31T12:51:35.359238Z","iopub.execute_input":"2023-01-31T12:51:35.359638Z","iopub.status.idle":"2023-01-31T12:51:35.400112Z","shell.execute_reply.started":"2023-01-31T12:51:35.359606Z","shell.execute_reply":"2023-01-31T12:51:35.398828Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission 1: only contains test data result in too many Missing values\n\nWe know that test data are truncated train data by time. It means that predictions I have now only captured part of the features. The result should be merged with ___?\n\n## Submission on the test data scoring in 0.061\nUpdate the missing values to the most frequenctly aid in the training dataset. ","metadata":{}},{"cell_type":"code","source":"\ni = 0\npreds = []\n\nwhile i< 3:\n\n    session = sample_train_df['events'][i]\n    #print(session[0])\n    # split on type\n\n    \n    # Saved aids for each type\n    click = []\n    carts = []\n    order = []\n    for d in session:\n        if d['type'] == 'clicks':\n            click.append(d['aid'])\n        if d['type'] == 'carts':\n            carts.append(d['aid'])\n        if d['type'] == 'orders':\n            order.append(d['aid'])\n    # next move addin counter\n    a = Counter(click)\n    top_click = nlargest(20, a, key = a.get)\n    b = Counter(carts)\n    top_carts = nlargest(20, b, key = b.get)\n    c = Counter(order)\n    top_order = nlargest(20, c, key = c.get)\n    print(f'topclick:{top_click},top cart{top_carts}, top order{top_order}')\n#  This is the full version for (str(id) for id in top_click)\n#     str(id) for id in top_click:\n#         print(str(id))\n    \n    \n\n# Append three types preds in sequence \n    preds.append(\" \".join([str(id) for id in top_click]))\n    preds.append(\" \".join([str(id) for id in top_carts]))\n    preds.append(\" \".join([str(id) for id in top_order]))\n    i = i+1","metadata":{"execution":{"iopub.status.busy":"2023-01-31T12:52:26.609096Z","iopub.execute_input":"2023-01-31T12:52:26.609521Z","iopub.status.idle":"2023-01-31T12:52:26.641119Z","shell.execute_reply.started":"2023-01-31T12:52:26.609489Z","shell.execute_reply":"2023-01-31T12:52:26.639854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check on the lenghth, should be 3X3\nlen(preds)","metadata":{"execution":{"iopub.status.busy":"2023-01-31T12:52:37.247303Z","iopub.execute_input":"2023-01-31T12:52:37.247735Z","iopub.status.idle":"2023-01-31T12:52:37.255236Z","shell.execute_reply.started":"2023-01-31T12:52:37.247700Z","shell.execute_reply":"2023-01-31T12:52:37.254008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The results shows even in Train data there are not always 20 recommandations to have\n# Need to figure out the source to fill in those blanks\n","metadata":{"execution":{"iopub.status.busy":"2023-01-31T08:03:39.314599Z","iopub.execute_input":"2023-01-31T08:03:39.314918Z","iopub.status.idle":"2023-01-31T08:03:39.335231Z","shell.execute_reply.started":"2023-01-31T08:03:39.314892Z","shell.execute_reply":"2023-01-31T08:03:39.334241Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prep","metadata":{}},{"cell_type":"code","source":"# #Save 3 types in different lists\n# click = []\n# carts = []\n# order = []\n# for d in session0:\n#     if d['type'] == 'clicks':\n#         click.append(d['aid'])\n#     if d['type'] == 'carts':\n#         carts.append(d['aid'])\n#     if d['type'] == 'orders':\n#         order.append(d['aid'])\n","metadata":{"execution":{"iopub.status.busy":"2023-01-31T04:09:23.260092Z","iopub.execute_input":"2023-01-31T04:09:23.260599Z","iopub.status.idle":"2023-01-31T04:09:23.267548Z","shell.execute_reply.started":"2023-01-31T04:09:23.260558Z","shell.execute_reply":"2023-01-31T04:09:23.266425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #top 1 frequent\n# from collections import Counter\n\n# a = Counter(click)\n\n \n# def most_frequent(List):\n#     occurence_count = Counter(List)\n#     return occurence_count.most_common(1)\n\n# most_frequent(a)","metadata":{"execution":{"iopub.status.busy":"2023-01-31T04:11:47.493298Z","iopub.execute_input":"2023-01-31T04:11:47.493703Z","iopub.status.idle":"2023-01-31T04:11:47.502297Z","shell.execute_reply.started":"2023-01-31T04:11:47.493671Z","shell.execute_reply":"2023-01-31T04:11:47.501224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #top 20 frequent\n# a = Counter(click)\n# top_click_article = nlargest(20, a, key = a.get)","metadata":{"execution":{"iopub.status.busy":"2023-01-31T04:13:05.234693Z","iopub.execute_input":"2023-01-31T04:13:05.235097Z","iopub.status.idle":"2023-01-31T04:13:05.239944Z","shell.execute_reply.started":"2023-01-31T04:13:05.235064Z","shell.execute_reply":"2023-01-31T04:13:05.238995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #timestamp \n# import datetime\n\n# timestamp = \"1659367439426\"\n# your_dt = datetime.datetime.fromtimestamp(int(timestamp)/1000)  # using the locauil timezone\n# print(your_dt.strftime(\"%Y-%m-%d %H:%M:%S\")) ","metadata":{"execution":{"iopub.status.busy":"2023-01-17T22:26:47.150478Z","iopub.execute_input":"2023-01-17T22:26:47.151024Z","iopub.status.idle":"2023-01-17T22:26:47.159112Z","shell.execute_reply.started":"2023-01-17T22:26:47.150965Z","shell.execute_reply":"2023-01-17T22:26:47.157626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission 2: fill the missing value with the most frequent Aid in training data","metadata":{}},{"cell_type":"code","source":"# Find out the top20 most frequent aids in each type\ni = 0\n\nclick = []\ncarts = []\norder = []\nwhile i< 1000:\n\n    session = sample_train_df['events'][i]\n    #print(session[0])\n    # split on type\n\n    \n    # Saved aids for each type\n\n    for d in session:\n        if d['type'] == 'clicks':\n            click.append(d['aid'])\n        if d['type'] == 'carts':\n            carts.append(d['aid'])\n        if d['type'] == 'orders':\n            order.append(d['aid'])\n    \n    i+=1 \n            ","metadata":{"execution":{"iopub.status.busy":"2023-01-31T12:57:20.707753Z","iopub.execute_input":"2023-01-31T12:57:20.708259Z","iopub.status.idle":"2023-01-31T12:57:20.796841Z","shell.execute_reply.started":"2023-01-31T12:57:20.708222Z","shell.execute_reply":"2023-01-31T12:57:20.795476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a_C = Counter(click)\ntop_click_C = nlargest(20, a_C, key = a_C.get)\nb_C = Counter(carts)\ntop_carts_C = nlargest(20, b_C, key = b_C.get)\nc_C = Counter(order)\ntop_order_C = nlargest(20, c_C, key = c_C.get)","metadata":{"execution":{"iopub.status.busy":"2023-01-31T12:57:23.339628Z","iopub.execute_input":"2023-01-31T12:57:23.340122Z","iopub.status.idle":"2023-01-31T12:57:23.378475Z","shell.execute_reply.started":"2023-01-31T12:57:23.340085Z","shell.execute_reply":"2023-01-31T12:57:23.377203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top_click_C","metadata":{"execution":{"iopub.status.busy":"2023-01-31T12:57:26.821695Z","iopub.execute_input":"2023-01-31T12:57:26.822126Z","iopub.status.idle":"2023-01-31T12:57:26.829664Z","shell.execute_reply.started":"2023-01-31T12:57:26.822090Z","shell.execute_reply":"2023-01-31T12:57:26.828646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check on the submission file","metadata":{}},{"cell_type":"code","source":"sample_submission = pd.read_csv(SAMPLE_SUB_PATH)\nsample_submission.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-31T13:30:15.746736Z","iopub.execute_input":"2023-01-31T13:30:15.747242Z","iopub.status.idle":"2023-01-31T13:30:22.318035Z","shell.execute_reply.started":"2023-01-31T13:30:15.747204Z","shell.execute_reply":"2023-01-31T13:30:22.316784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-01-31T13:30:32.174727Z","iopub.execute_input":"2023-01-31T13:30:32.176387Z","iopub.status.idle":"2023-01-31T13:30:32.184650Z","shell.execute_reply.started":"2023-01-31T13:30:32.176321Z","shell.execute_reply":"2023-01-31T13:30:32.183416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission['session_type'].values[-1]","metadata":{"execution":{"iopub.status.busy":"2023-01-31T13:30:34.321195Z","iopub.execute_input":"2023-01-31T13:30:34.321650Z","iopub.status.idle":"2023-01-31T13:30:34.330061Z","shell.execute_reply.started":"2023-01-31T13:30:34.321615Z","shell.execute_reply":"2023-01-31T13:30:34.328837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Applying on Test data","metadata":{}},{"cell_type":"code","source":"# read test data 1671803 data for this submission\nsample_size = 1671803\n\nchunks = pd.read_json(TEST_PATH, lines=True, chunksize = sample_size)\n\nfor c in chunks:\n    sample_test_df = c\n    break\n","metadata":{"execution":{"iopub.status.busy":"2023-01-31T13:20:51.113758Z","iopub.execute_input":"2023-01-31T13:20:51.114248Z","iopub.status.idle":"2023-01-31T13:21:22.733518Z","shell.execute_reply.started":"2023-01-31T13:20:51.114201Z","shell.execute_reply":"2023-01-31T13:21:22.732397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_test_df['events'][1671802]","metadata":{"execution":{"iopub.status.busy":"2023-01-31T13:25:45.426365Z","iopub.execute_input":"2023-01-31T13:25:45.426825Z","iopub.status.idle":"2023-01-31T13:25:45.435286Z","shell.execute_reply.started":"2023-01-31T13:25:45.426786Z","shell.execute_reply":"2023-01-31T13:25:45.433813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Copy pasting . . .\n\n\ni = 0\npreds = []\n# in total 1671803 rows\nwhile i<  1671803:\n\n    session = sample_test_df['events'][i]\n \n    click = []\n    carts = []\n    order = []\n    for d in session:\n        if d['type'] == 'clicks':\n            click.append(d['aid'])\n        elif  d['type'] == 'carts':\n            carts.append(d['aid'])\n        elif  d['type'] == 'orders':\n            order.append(d['aid'])\n   \n    a = Counter(click)\n    top_click = nlargest(20, a, key = a.get)\n    b = Counter(carts)\n    top_carts = nlargest(20, b, key = b.get)\n    c = Counter(order)\n    top_order = nlargest(20, c, key = c.get)\n    \n    #append\n    \n    if  len(top_click) == 20 :\n        preds.append(\" \".join([str(id) for id in top_click]))\n    \n    elif len(top_click) < 20 :\n        \n        fillin_click = sample(top_click_C,20-len(top_click))\n        \n        top_click.extend(fillin_click)\n        preds.append(\" \".join([str(id) for id in top_click]))\n\n        \n    if  len(top_carts) == 20 :\n        preds.append(\" \".join([str(id) for id in top_carts]))\n        \n    elif len(top_carts) < 20 :\n        \n        fillin_carts = sample(top_carts_C,20-len(top_carts))\n        \n        top_carts.extend(fillin_carts)\n        preds.append(\" \".join([str(id) for id in top_carts]))                          \n\n                      \n    if  len(top_order) == 20 :\n        preds.append(\" \".join([str(id) for id in top_order]))\n\n    elif len(top_order) < 20 :\n        fillin_order = sample(top_order_C,20-len(top_order))\n        \n        top_order.extend(fillin_order)\n        preds.append(\" \".join([str(id) for id in top_order]))     \n        \n                              \n    i = i+1","metadata":{"execution":{"iopub.status.busy":"2023-01-31T13:21:39.979759Z","iopub.execute_input":"2023-01-31T13:21:39.980193Z","iopub.status.idle":"2023-01-31T13:24:40.324515Z","shell.execute_reply.started":"2023-01-31T13:21:39.980159Z","shell.execute_reply":"2023-01-31T13:24:40.323193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#preds","metadata":{"execution":{"iopub.status.busy":"2023-01-31T12:45:52.630820Z","iopub.execute_input":"2023-01-31T12:45:52.631396Z","iopub.status.idle":"2023-01-31T12:45:52.639644Z","shell.execute_reply.started":"2023-01-31T12:45:52.631347Z","shell.execute_reply":"2023-01-31T12:45:52.638415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check on the len 3times of the input test data\nlen(preds)","metadata":{"execution":{"iopub.status.busy":"2023-01-31T13:30:55.720169Z","iopub.execute_input":"2023-01-31T13:30:55.721026Z","iopub.status.idle":"2023-01-31T13:30:55.730074Z","shell.execute_reply.started":"2023-01-31T13:30:55.720972Z","shell.execute_reply":"2023-01-31T13:30:55.728584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission['labels'] = preds","metadata":{"execution":{"iopub.status.busy":"2023-01-31T13:30:58.024659Z","iopub.execute_input":"2023-01-31T13:30:58.025100Z","iopub.status.idle":"2023-01-31T13:30:58.798947Z","shell.execute_reply.started":"2023-01-31T13:30:58.025066Z","shell.execute_reply":"2023-01-31T13:30:58.797446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-31T13:31:31.360790Z","iopub.execute_input":"2023-01-31T13:31:31.361289Z","iopub.status.idle":"2023-01-31T13:31:52.814076Z","shell.execute_reply.started":"2023-01-31T13:31:31.361252Z","shell.execute_reply":"2023-01-31T13:31:52.812994Z"},"trusted":true},"execution_count":null,"outputs":[]}]}