{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-17T21:19:37.166671Z","iopub.execute_input":"2022-12-17T21:19:37.167066Z","iopub.status.idle":"2022-12-17T21:19:37.175334Z","shell.execute_reply.started":"2022-12-17T21:19:37.167036Z","shell.execute_reply":"2022-12-17T21:19:37.174225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom collections import defaultdict\nimport gc\nfrom operator import itemgetter\nimport random","metadata":{"execution":{"iopub.status.busy":"2022-12-17T21:44:10.352875Z","iopub.execute_input":"2022-12-17T21:44:10.353305Z","iopub.status.idle":"2022-12-17T21:44:10.359519Z","shell.execute_reply.started":"2022-12-17T21:44:10.353261Z","shell.execute_reply":"2022-12-17T21:44:10.358084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# sample_size = 1000\n\n# chunks = pd.read_json(\"/kaggle/input/otto-recommender-system/train.jsonl\" , lines=True, chunksize = sample_size)","metadata":{"execution":{"iopub.status.busy":"2022-12-17T21:44:18.780251Z","iopub.execute_input":"2022-12-17T21:44:18.780652Z","iopub.status.idle":"2022-12-17T21:44:18.788569Z","shell.execute_reply.started":"2022-12-17T21:44:18.780620Z","shell.execute_reply":"2022-12-17T21:44:18.787341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def loadData(path = \"/kaggle/input/otto-recommender-system/train.jsonl\", sizePerChunk = 1000, truncated = True):\n    \"\"\"\n    Read the data and explode the data into the schema of \n        [session, aid, ts, type] with unique index. \n    \"\"\"\n    chunks = pd.read_json(path, lines = True, chunksize = sizePerChunk)\n    explodedDf = pd.DataFrame()\n    \n    for chunk in chunks:\n        eventDict = {\"session\": [], \"aid\": [], \"ts\": [], \"type\": []}\n        \n        for session, events in zip(chunk[\"session\"].tolist(), chunk[\"events\"].tolist()):\n            for event in events: \n                eventDict[\"session\"].append(session)\n                eventDict[\"aid\"].append(event[\"aid\"])\n                eventDict[\"ts\"].append(event[\"ts\"])\n                eventDict[\"type\"].append(event[\"type\"])\n        \n        chunkSession = pd.DataFrame(eventDict)\n        explodedDf = pd.concat([explodedDf, chunkSession])\n        \n        if truncated:\n            break\n    \n    return explodedDf\n\ndef dataPreprocessing(explodedDf):\n    \"\"\"\n    convert the explodedDf data in dictData format, described in getItemSimAndLikeMatrix method. \n    \"\"\"\n    interactDf = explodedDf[['session','aid']]\n    listData= []\n#     random.seed(3)\n    for idx, row in interactDf.iterrows():\n        user = int(row['session'])\n        item = int(row['aid'])\n        listData.append([user, item])\n            \n    dictData = dict()\n    for user, item in listData:\n        dictData.setdefault(user, set())\n        dictData[user].add(item)\n    \n    return dictData\n\ndef getItemSimAndLikeMatrix(dictData):\n    \"\"\"\n    dictData: User-Item table, in the shape of \n             {\n              session1: {aid1, aid2, ..... }\n              session2: {aid4, aid9, ..... },\n              .....\n             }\n    \"\"\"\n    N = defaultdict(int)  ## default a matrx/dict recording how many ppl/item likes the item, all item default at 0\n    itemSimMatrix = defaultdict(int) ## initiates the itemSimMatrix ->  {itemId_1 : {itemId_X: numSesLikedBoth, ....}}\n    for session, userAids in dictData.items():\n        for aid in userAids:\n            itemSimMatrix.setdefault(aid, dict())  ## set up the account for aid \n            N[aid] += 1\n            for aidReference in userAids:\n                if aid == aidReference:\n                    continue\n                itemSimMatrix[aid].setdefault(aidReference, 0)  ## initiate the count of aid - aidReference as 0 if it hasn't exists\n                itemSimMatrix[aid][aidReference] += 1           ## increment the similiary score 1 per mutual interaction\n                \n    return N, itemSimMatrix\n        \n        \ndef recommend(dictData, userId, itemSimMatrix, N, K):\n    \"\"\"\n    user: userId to recommend\n    N: # of items to recommend\n    K: # of similar sessions(user) to look up\n    \"\"\"\n    recommends = dict()\n    items = dictData[userId] ## all items(aids) the user liked \n    for item in items:\n        ## look up the top liked score in {itemId_1 : {itemId_X: numSesLikedBoth, ....}} where item is set as itemId_1 here. \n        if itemSimMatrix[item] == 0:\n            continue\n        for i, sim, in sorted(itemSimMatrix[item].items(), key=itemgetter(1), reverse=True)[:K]: \n            if i == item:\n                continue  ## if the item is already in liked items, skip\n            recommends.setdefault(i, 0)   ## otherwise initiate the total score of recommendation of this item(itemId_X), 0 if it hasn't appear in the list before\n            recommends[i] += sim   ## then add the sim scores, in the naive version sim = 1, for more complicated ones, sim will be differed.\n            \n    return dict(sorted(recommends.items(), key=itemgetter(0), reverse=True)[:N])\n     \n# def convertStr(dictResult):\n#     return \" \".join([str(x) for x in recommend(test_dict_data, 12899782, item_sim_matrix, 20, 40).keys()])\n                ","metadata":{"execution":{"iopub.status.busy":"2022-12-17T22:46:50.594750Z","iopub.execute_input":"2022-12-17T22:46:50.595140Z","iopub.status.idle":"2022-12-17T22:46:50.615478Z","shell.execute_reply.started":"2022-12-17T22:46:50.595106Z","shell.execute_reply":"2022-12-17T22:46:50.613794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_dict_data = dataPreprocessing(loadData(sizePerChunk = 100000))\nitem_like_matrix, item_sim_matrix = getItemSimAndLikeMatrix(train_dict_data)","metadata":{"execution":{"iopub.status.busy":"2022-12-17T22:47:18.907821Z","iopub.execute_input":"2022-12-17T22:47:18.908228Z","iopub.status.idle":"2022-12-17T22:54:29.417047Z","shell.execute_reply.started":"2022-12-17T22:47:18.908196Z","shell.execute_reply":"2022-12-17T22:54:29.415936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntest_dict_data = dataPreprocessing(loadData(path =\"/kaggle/input/otto-recommender-system/test.jsonl\", sizePerChunk = 10000, truncated=False))\n#test_dict_data\nrecommend(test_dict_data, 12899782, item_sim_matrix, 20, 40)","metadata":{"execution":{"iopub.status.busy":"2022-12-17T23:08:08.385047Z","iopub.execute_input":"2022-12-17T23:08:08.385511Z","iopub.status.idle":"2022-12-17T23:15:37.289889Z","shell.execute_reply.started":"2022-12-17T23:08:08.385473Z","shell.execute_reply":"2022-12-17T23:15:37.288718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nsession = [x for x in test_dict_data.keys()]","metadata":{"execution":{"iopub.status.busy":"2022-12-17T23:34:35.172269Z","iopub.execute_input":"2022-12-17T23:34:35.172882Z","iopub.status.idle":"2022-12-17T23:34:35.303205Z","shell.execute_reply.started":"2022-12-17T23:34:35.172825Z","shell.execute_reply":"2022-12-17T23:34:35.302123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# recs = []\n# for i in session:\n#     single_rec = \" \".join([str(x) for x in recommend(test_dict_data, i, item_sim_matrix, 20, 20).keys()])\n#     recs.append(single_rec)","metadata":{"execution":{"iopub.status.busy":"2022-12-18T01:52:32.438927Z","iopub.execute_input":"2022-12-18T01:52:32.439629Z","iopub.status.idle":"2022-12-18T01:52:32.444432Z","shell.execute_reply.started":"2022-12-18T01:52:32.439595Z","shell.execute_reply":"2022-12-18T01:52:32.443425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntest_sub = pd.DataFrame(list(zip(session)),\n               columns =['session_key'])","metadata":{"execution":{"iopub.status.busy":"2022-12-18T00:00:38.514136Z","iopub.execute_input":"2022-12-18T00:00:38.514550Z","iopub.status.idle":"2022-12-18T00:00:39.397553Z","shell.execute_reply.started":"2022-12-18T00:00:38.514515Z","shell.execute_reply":"2022-12-18T00:00:39.396433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntest_sub[\"aid_rec\"] = test_sub.apply(lambda row: \" \".join([str(i) for i in recommend(test_dict_data, row[\"session_key\"], item_sim_matrix, 20, 20).keys()]), axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-12-18T00:16:34.733367Z","iopub.execute_input":"2022-12-18T00:16:34.733783Z","iopub.status.idle":"2022-12-18T01:20:23.632314Z","shell.execute_reply.started":"2022-12-18T00:16:34.733748Z","shell.execute_reply":"2022-12-18T01:20:23.630965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_sub","metadata":{"execution":{"iopub.status.busy":"2022-12-18T01:27:51.350548Z","iopub.execute_input":"2022-12-18T01:27:51.351183Z","iopub.status.idle":"2022-12-18T01:27:51.365871Z","shell.execute_reply.started":"2022-12-18T01:27:51.351139Z","shell.execute_reply":"2022-12-18T01:27:51.364760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub = pd.read_csv('/kaggle/input/otto-recommender-system/sample_submission.csv')\nsample_sub.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-12-18T01:41:40.242922Z","iopub.execute_input":"2022-12-18T01:41:40.243369Z","iopub.status.idle":"2022-12-18T01:41:46.298260Z","shell.execute_reply.started":"2022-12-18T01:41:40.243332Z","shell.execute_reply":"2022-12-18T01:41:46.295968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nsample_sub[\"session_key\"] = sample_sub.apply(lambda row: int(row[\"session_type\"].split(\"_\")[0]), axis=1)\n#[\"session_type\"].str.split(\"_\")[0]#.apply(lambda row: int(row[\"session_type\"].split(\"_\")[0]), axis = 1)\nsample_sub","metadata":{"execution":{"iopub.status.busy":"2022-12-18T01:51:36.312606Z","iopub.execute_input":"2022-12-18T01:51:36.313449Z","iopub.status.idle":"2022-12-18T01:52:32.437153Z","shell.execute_reply.started":"2022-12-18T01:51:36.313400Z","shell.execute_reply":"2022-12-18T01:52:32.436434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n## get the best seller list\ntrain_df = loadData(sizePerChunk = 100000)\nbest_sellers = train_df[train_df.type == \"orders\"].groupby([\"aid\", \"type\"])[\"session\"].count().reset_index()\nbest_sellers.columns = [\"aid\", \"type\", \"counts\"]\nbest_sellers = best_sellers.sort_values(['counts'], ascending=False).reset_index()#.head(10)\nbest_sold_str =  ' '.join(best_sellers[:20].aid.astype(\"str\").tolist())\nbest_sold_str","metadata":{"execution":{"iopub.status.busy":"2022-12-18T01:58:25.841890Z","iopub.execute_input":"2022-12-18T01:58:25.842291Z","iopub.status.idle":"2022-12-18T01:58:44.106846Z","shell.execute_reply.started":"2022-12-18T01:58:25.842258Z","shell.execute_reply":"2022-12-18T01:58:44.106069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nsub = pd.merge(sample_sub, test_sub, on = \"session_key\", how='left')\nsub = sub[[\"session_type\", \"aid_rec\"]]\nsub.columns = [\"session_type\", \"labels\"]\nsub[\"labels\"] = sub.apply(lambda row: row[\"labels\"] if len(row[\"labels\"]) else best_sold_str, axis=1)\nsub","metadata":{"execution":{"iopub.status.busy":"2022-12-18T02:06:56.777365Z","iopub.execute_input":"2022-12-18T02:06:56.778483Z","iopub.status.idle":"2022-12-18T02:08:11.379320Z","shell.execute_reply.started":"2022-12-18T02:06:56.778444Z","shell.execute_reply":"2022-12-18T02:08:11.378122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-18T02:00:52.692087Z","iopub.execute_input":"2022-12-18T02:00:52.692492Z","iopub.status.idle":"2022-12-18T02:00:52.701003Z","shell.execute_reply.started":"2022-12-18T02:00:52.692460Z","shell.execute_reply":"2022-12-18T02:00:52.699870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}