{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is this notebook?\n- Almost same as [this Mr. Tawara's great notebook](https://www.kaggle.com/code/ttahara/otto-mors-aid-frequency-baseline/notebook?scriptVersionId=109781928)\n  - **Before check my notebook, please read the notebook please!**\n- The problem in the notebook is that sessions in test data is really short\n    - Sometimes, shorter than 20\n    - It's not good for  Recall@20\n- Therefore, I use co-occurred a-id with test session a-id\n    - use them for padding\n    - eg. \n- convert .json to dataframe, I used [this Mr. COLUM2131's great notebook](https://www.kaggle.com/code/columbia2131/otto-read-a-chunk-of-jsonl-to-manageable-df)\n    - **Before check my notebook, please also read the notebook please!**","metadata":{}},{"cell_type":"markdown","source":"# reading data","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport datetime\n\n\n\nfrom pathlib import Path\n\ndata_path = Path('/kaggle/input/otto-recommender-system/')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-03T08:28:22.353313Z","iopub.execute_input":"2022-11-03T08:28:22.354707Z","iopub.status.idle":"2022-11-03T08:28:22.368364Z","shell.execute_reply.started":"2022-11-03T08:28:22.354662Z","shell.execute_reply":"2022-11-03T08:28:22.366745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## json to df ( I used [this Mr. COLUM2131's great notebook](https://www.kaggle.com/code/columbia2131/otto-read-a-chunk-of-jsonl-to-manageable-df))","metadata":{}},{"cell_type":"code","source":"def json_to_df (json_file_name):\n    train_sessions = pd.DataFrame()\n    chunks = pd.read_json(data_path / json_file_name, lines=True, chunksize=100_000)\n\n    for e, chunk in enumerate(chunks):\n        event_dict = {\n            'session': [],\n            'aid': [],\n            'ts': [],\n            'type': [],\n        }\n        if e < 2:\n            # train_sessions = pd.concat([train_sessions, chunk])\n            for session, events in zip(chunk['session'].tolist(), chunk['events'].tolist()):\n                for event in events:\n                    event_dict['session'].append(session)\n                    event_dict['aid'].append(event['aid'])\n                    event_dict['ts'].append(event['ts'])\n                    event_dict['type'].append(event['type'])\n            chunk_session = pd.DataFrame(event_dict)\n            train_sessions = pd.concat([train_sessions, chunk_session])\n        else:\n            break\n\n    train_sessions = train_sessions.reset_index(drop=True)\n    return train_sessions","metadata":{"execution":{"iopub.status.busy":"2022-11-03T08:28:22.370295Z","iopub.execute_input":"2022-11-03T08:28:22.371428Z","iopub.status.idle":"2022-11-03T08:28:22.384749Z","shell.execute_reply.started":"2022-11-03T08:28:22.371384Z","shell.execute_reply":"2022-11-03T08:28:22.383359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_sessions =json_to_df (\"train.jsonl\")\ntest_sessions =json_to_df (\"test.jsonl\")\ntrain_sessions[\"ts\"] /= 1000\ntest_sessions[\"ts\"] /= 1000","metadata":{"execution":{"iopub.status.busy":"2022-11-03T08:28:22.386274Z","iopub.execute_input":"2022-11-03T08:28:22.387301Z","iopub.status.idle":"2022-11-03T08:29:26.390342Z","shell.execute_reply.started":"2022-11-03T08:28:22.387262Z","shell.execute_reply":"2022-11-03T08:29:26.388679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# to make co-occur, drop duplicates","metadata":{}},{"cell_type":"code","source":"removed_duplicated_train = train_sessions.drop_duplicates([\"session\",\"aid\"])[[\"session\",\"aid\"]]\nremoved_duplicated_test = test_sessions.drop_duplicates([\"session\",\"aid\"])[[\"session\",\"aid\"]]","metadata":{"execution":{"iopub.status.busy":"2022-11-03T08:29:26.393303Z","iopub.execute_input":"2022-11-03T08:29:26.393747Z","iopub.status.idle":"2022-11-03T08:29:29.513967Z","shell.execute_reply.started":"2022-11-03T08:29:26.393711Z","shell.execute_reply":"2022-11-03T08:29:29.512268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_sessions\nimport gc\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-11-03T08:29:29.515829Z","iopub.execute_input":"2022-11-03T08:29:29.516260Z","iopub.status.idle":"2022-11-03T08:29:32.712734Z","shell.execute_reply.started":"2022-11-03T08:29:29.516225Z","shell.execute_reply":"2022-11-03T08:29:32.711510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"aid_set_test = set(removed_duplicated_test[\"aid\"].unique())","metadata":{"execution":{"iopub.status.busy":"2022-11-03T08:29:32.713850Z","iopub.execute_input":"2022-11-03T08:29:32.714180Z","iopub.status.idle":"2022-11-03T08:29:32.805610Z","shell.execute_reply.started":"2022-11-03T08:29:32.714149Z","shell.execute_reply":"2022-11-03T08:29:32.804243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# make co-occur (if you know better or more efficient methods, Please comment!)","metadata":{}},{"cell_type":"code","source":"from tqdm import tqdm\nsize = 1000\nfrom collections import defaultdict\ndef int_f():\n    return defaultdict(int)\n\n# make co-occur dict\ntrain_co_occur_count = defaultdict(int_f)\n\nfor i in tqdm(range(0,200000,size)):\n    f1 =  i <=removed_duplicated_train[\"session\"]\n    f2 = removed_duplicated_train[\"session\"] < i + size\n    partial_train = removed_duplicated_train[f1 & f2]\n    temp = pd.merge(partial_train,partial_train, on = [\"session\"])\n    \n    \n    # For memory problem, only set value for aid in testdata\n    f1 = temp[\"aid_x\"].isin(aid_set_test)\n    f2 = temp[\"aid_y\"].isin(aid_set_test)\n    temp = temp[f1 | f2]\n    \n    temp =temp[[\"aid_x\",\"aid_y\"]].value_counts().reset_index()\n    for x,y,co_count in  zip(temp[\"aid_x\"],temp[\"aid_y\"],temp[0]):\n        if x in aid_set_test:\n            train_co_occur_count[x][y] += co_count\n        if y in aid_set_test:\n            train_co_occur_count[y][x] += co_count","metadata":{"execution":{"iopub.status.busy":"2022-11-03T08:29:32.807751Z","iopub.execute_input":"2022-11-03T08:29:32.808165Z","iopub.status.idle":"2022-11-03T08:56:01.008575Z","shell.execute_reply.started":"2022-11-03T08:29:32.808128Z","shell.execute_reply":"2022-11-03T08:56:01.007069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# sort with how many co-occur (to save memory)","metadata":{}},{"cell_type":"code","source":"for aid in aid_set_test:\n    train_co_occur_count[aid] = sorted(train_co_occur_count[aid].items(),key = lambda x:-x[1])[:20]\n\nwith open('train_co_occur_count.pickle', 'wb') as fi:\n    pickle.dump(dict(train_co_occur_count), fi)","metadata":{"execution":{"iopub.status.busy":"2022-11-03T08:57:59.369413Z","iopub.execute_input":"2022-11-03T08:57:59.370155Z","iopub.status.idle":"2022-11-03T09:00:13.044867Z","shell.execute_reply.started":"2022-11-03T08:57:59.370111Z","shell.execute_reply":"2022-11-03T09:00:13.043360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict with method of [Mr. Tawara's great notebook](https://www.kaggle.com/code/ttahara/otto-mors-aid-frequency-baseline/notebook?scriptVersionId=109781928)","metadata":{}},{"cell_type":"code","source":"from collections import Counter\n","metadata":{"execution":{"iopub.status.busy":"2022-11-03T09:50:42.064224Z","iopub.execute_input":"2022-11-03T09:50:42.065119Z","iopub.status.idle":"2022-11-03T09:50:42.070087Z","shell.execute_reply.started":"2022-11-03T09:50:42.065079Z","shell.execute_reply":"2022-11-03T09:50:42.068331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sorted_ids_list = []\ntest_sessions = pd.read_json(\"../input/otto-recommender-system/test.jsonl\", lines=True, chunksize=100000)\n\nfor chunk in tqdm(test_sessions):\n    \n    for session_id, events in chunk.values:\n        aid_list = []\n        for action in events:\n            aid_list.append(action[\"aid\"])\n\n        cnt = Counter(aid_list)\n        sorted_aids = sorted(set(aid_list), key=lambda x: cnt[x], reverse=True)\n        sorted_ids_list.append([session_id, sorted_aids])","metadata":{"execution":{"iopub.status.busy":"2022-11-03T09:50:42.783968Z","iopub.execute_input":"2022-11-03T09:50:42.784415Z","iopub.status.idle":"2022-11-03T09:51:54.652132Z","shell.execute_reply.started":"2022-11-03T09:50:42.784382Z","shell.execute_reply":"2022-11-03T09:51:54.650549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_list = []\n\n# length of prediction\nlen_list = []\nfor session_id, sorted_aids in sorted_ids_list:\n    sorted_aids_20_str = \" \".join(map(str, sorted_aids[:20]))\n    for _ in range(3):\n        len_list.append(len(sorted_aids[:20]))\n    data_list.append([f\"{session_id}_clicks\", sorted_aids_20_str])\n    data_list.append([f\"{session_id}_carts\", sorted_aids_20_str])\n    data_list.append([f\"{session_id}_orders\", sorted_aids_20_str])","metadata":{"execution":{"iopub.status.busy":"2022-11-03T10:44:37.125066Z","iopub.execute_input":"2022-11-03T10:44:37.125470Z","iopub.status.idle":"2022-11-03T10:44:43.765558Z","shell.execute_reply.started":"2022-11-03T10:44:37.125436Z","shell.execute_reply":"2022-11-03T10:44:43.763626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame(\n    data_list, columns=[\"session_type\", \"labels\"])","metadata":{"execution":{"iopub.status.busy":"2022-11-03T09:55:35.470573Z","iopub.execute_input":"2022-11-03T09:55:35.471015Z","iopub.status.idle":"2022-11-03T09:55:36.726643Z","shell.execute_reply.started":"2022-11-03T09:55:35.470983Z","shell.execute_reply":"2022-11-03T09:55:36.725401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# padding with co-occur","metadata":{}},{"cell_type":"code","source":"def padding(row):\n    \n    padding_labels = []\n    labels = list(map(int,row[\"labels\"].split(\" \")))\n    used_label = set(labels)\n    \n    # shortage_label_size → how many padding is needed \n    shortage_label_size = 20 - row[\"len\"]\n    \n    \n    for label in labels:\n        co_occur = train_co_occur_count[label]\n        if shortage_label_size <= 0:\n            break\n        for new_label in co_occur:\n            if new_label[0] not in used_label:\n                padding_labels.append(new_label[0])\n                shortage_label_size -= 1\n            if shortage_label_size <= 0:\n                break\n        if shortage_label_size <= 0:\n            break\n    \n    return \" \".join(map(str,padding_labels))","metadata":{"execution":{"iopub.status.busy":"2022-11-03T10:09:12.581569Z","iopub.execute_input":"2022-11-03T10:09:12.581975Z","iopub.status.idle":"2022-11-03T10:09:12.590639Z","shell.execute_reply.started":"2022-11-03T10:09:12.581942Z","shell.execute_reply":"2022-11-03T10:09:12.589309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub[\"len\"] = len_list\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub[\"cooccur_predict\"] = sub.apply(padding,axis = 1)\nsub[\"labels\"] = sub[\"labels\"] + \" \" + sub[\"cooccur_predict\"]","metadata":{"execution":{"iopub.status.busy":"2022-11-03T10:09:47.157121Z","iopub.execute_input":"2022-11-03T10:09:47.157851Z","iopub.status.idle":"2022-11-03T10:12:00.876511Z","shell.execute_reply.started":"2022-11-03T10:09:47.157815Z","shell.execute_reply":"2022-11-03T10:12:00.875403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub[[\"session_type\", \"labels\"]].to_csv(\"submission.csv\", index=False)\n","metadata":{"execution":{"iopub.status.busy":"2022-11-03T10:13:39.051378Z","iopub.execute_input":"2022-11-03T10:13:39.051755Z","iopub.status.idle":"2022-11-03T10:13:59.503618Z","shell.execute_reply.started":"2022-11-03T10:13:39.051726Z","shell.execute_reply":"2022-11-03T10:13:59.502729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub[[\"session_type\", \"labels\"]].head()","metadata":{"execution":{"iopub.status.busy":"2022-11-03T10:14:07.236834Z","iopub.execute_input":"2022-11-03T10:14:07.238187Z","iopub.status.idle":"2022-11-03T10:14:07.255119Z","shell.execute_reply.started":"2022-11-03T10:14:07.238147Z","shell.execute_reply":"2022-11-03T10:14:07.253670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}