{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import json\nimport pandas as pd\nimport numpy as np\nfrom collections import defaultdict\nfrom tqdm import tqdm_notebook\nimport gc","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-16T18:58:09.979844Z","iopub.execute_input":"2022-11-16T18:58:09.980307Z","iopub.status.idle":"2022-11-16T18:58:10.004325Z","shell.execute_reply.started":"2022-11-16T18:58:09.980209Z","shell.execute_reply":"2022-11-16T18:58:10.003300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n!pip install fastparquet","metadata":{"execution":{"iopub.status.busy":"2022-11-16T18:58:10.030465Z","iopub.execute_input":"2022-11-16T18:58:10.031124Z","iopub.status.idle":"2022-11-16T18:58:25.552872Z","shell.execute_reply.started":"2022-11-16T18:58:10.031088Z","shell.execute_reply":"2022-11-16T18:58:25.551355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N_train = 216716096\nN_test = 6928123","metadata":{"execution":{"iopub.status.busy":"2022-11-16T18:58:25.555195Z","iopub.execute_input":"2022-11-16T18:58:25.555563Z","iopub.status.idle":"2022-11-16T18:58:25.560504Z","shell.execute_reply.started":"2022-11-16T18:58:25.555528Z","shell.execute_reply":"2022-11-16T18:58:25.559364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_path = '../input/otto-recommender-system/train.jsonl'\ntest_data_path = '../input/otto-recommender-system/test.jsonl'","metadata":{"execution":{"iopub.status.busy":"2022-11-16T18:58:25.561989Z","iopub.execute_input":"2022-11-16T18:58:25.562986Z","iopub.status.idle":"2022-11-16T18:58:25.571982Z","shell.execute_reply.started":"2022-11-16T18:58:25.562951Z","shell.execute_reply":"2022-11-16T18:58:25.570972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def otto_to_df(data_path, N):\n    type2id = {'clicks': 0, 'carts': 1, 'orders': 2}\n    \n    session_arr = np.zeros(N, dtype = np.uint32)\n    aid_arr = np.zeros(N, dtype = np.uint32)\n    ts_arr = np.zeros(N, dtype = np.uint64)\n    type_arr = np.zeros(N, dtype = np.uint8)\n    \n    with open(data_path, 'r') as f:\n        row_count = 0\n        for line in f:\n            session = json.loads(line)\n\n            session_id = session['session']\n\n            for event in session['events']:\n                session_arr[row_count] = session_id\n                aid_arr[row_count] = event['aid']\n                ts_arr[row_count] = event['ts']\n                type_arr[row_count] = type2id[event['type']]\n                row_count += 1\n    \n    data = pd.DataFrame({'session': session_arr, \n                         'aid': aid_arr, \n                         'ts': ts_arr, \n                         'type': type_arr})\n    \n    return data\n                            \n  ","metadata":{"execution":{"iopub.status.busy":"2022-11-16T18:58:25.574314Z","iopub.execute_input":"2022-11-16T18:58:25.574634Z","iopub.status.idle":"2022-11-16T18:58:25.591717Z","shell.execute_reply.started":"2022-11-16T18:58:25.574605Z","shell.execute_reply":"2022-11-16T18:58:25.590455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"  \ndef train_test_split(df, weeks = 1, frac = 0.01):\n    test_duration = weeks * 7*24*60*60*1000\n    split_ts = df.ts.max() - test_duration\n    \n    train = df.loc[df.ts <= split_ts]\n    test = df.loc[df.ts > split_ts]\n    \n    common_sessions = set(train.session).intersection(set(test.session))\n    \n    test = test.loc[~test.session.isin(common_sessions)]\n    \n    lucky_sessions = test.session.drop_duplicates().sample(frac = frac)\n    test = test.loc[test.session.isin(lucky_sessions)]\n    \n    new_test = []\n    new_val = []\n    \n    \n    for i, (_, g) in enumerate(test.groupby('session')):\n        if g.shape[0] == 1:\n            continue\n        cutoff = np.random.randint(1, g.shape[0])\n        \n        new_test.append(g.iloc[:cutoff])\n        new_val.append(g.iloc[cutoff:])\n        \n        #if i % 100000 == 0:\n        #    gc.collect()\n    \n    test = pd.concat(new_test).reset_index(drop = True)\n    labels = pd.concat(new_val).reset_index(drop = True)\n    \n    return train, test, labels\n    \n\n      \n    ","metadata":{"execution":{"iopub.status.busy":"2022-11-16T18:58:25.593120Z","iopub.execute_input":"2022-11-16T18:58:25.593429Z","iopub.status.idle":"2022-11-16T18:58:25.605697Z","shell.execute_reply.started":"2022-11-16T18:58:25.593401Z","shell.execute_reply":"2022-11-16T18:58:25.604895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = otto_to_df(train_data_path, N_train)\ntest = otto_to_df(test_data_path, N_test)","metadata":{"execution":{"iopub.status.busy":"2022-11-16T18:58:25.607167Z","iopub.execute_input":"2022-11-16T18:58:25.607836Z","iopub.status.idle":"2022-11-16T19:05:10.879697Z","shell.execute_reply.started":"2022-11-16T18:58:25.607795Z","shell.execute_reply":"2022-11-16T19:05:10.877548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.to_parquet('train', engine = 'fastparquet', index = False)\ntest.to_parquet('test', engine = 'fastparquet', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-11-16T19:05:10.882393Z","iopub.execute_input":"2022-11-16T19:05:10.882816Z","iopub.status.idle":"2022-11-16T19:05:10.888406Z","shell.execute_reply.started":"2022-11-16T19:05:10.882772Z","shell.execute_reply":"2022-11-16T19:05:10.887661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_train, _test, _val = train_test_split(train, weeks = 1, frac = 0.01)","metadata":{"execution":{"iopub.status.busy":"2022-11-16T19:05:10.889584Z","iopub.execute_input":"2022-11-16T19:05:10.890369Z","iopub.status.idle":"2022-11-16T19:06:10.293424Z","shell.execute_reply.started":"2022-11-16T19:05:10.890323Z","shell.execute_reply":"2022-11-16T19:06:10.292182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_train.to_parquet('_train', engine = 'fastparquet', index = False)\n_test.to_parquet('_test', engine = 'fastparquet', index = False)\n_val.to_parquet('_val', engine = 'fastparquet', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-11-16T19:06:10.303686Z","iopub.execute_input":"2022-11-16T19:06:10.304081Z","iopub.status.idle":"2022-11-16T19:06:54.197081Z","shell.execute_reply.started":"2022-11-16T19:06:10.304048Z","shell.execute_reply":"2022-11-16T19:06:54.195221Z"},"trusted":true},"execution_count":null,"outputs":[]}]}