{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Description\n- V4: base\n- V5: cut off the first week \n- V8: ~V16 validation, add item features and re-chunking\n- V10: new pipeline (candidates from notebook covi) 5 chunks\n- V11: ~V10 10chunks\n- V13: ~V11 test gpu\n- V15: ~V11 change aids3+aids2\n- V16: 150 cands\n- V17: 50 cands\n- V18: add timestamp features (out of memory)\n- V19: ~V15, only timestamp day features\n- V21: quantize V19 | 10 chunks\n- V23: ~V21 20 chunks\n- V25: 37 features \n- V27: ~V25, add common test aid","metadata":{}},{"cell_type":"code","source":"!pip install pyarrow fastparquet","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport time\nimport datetime\nfrom tqdm.notebook import tqdm\nimport os, sys, pickle, glob, gc\nfrom collections import Counter\nimport itertools\n\n# from multiprocessing import Pool\n# import psutil\n# N_CPU = psutil.cpu_count()\n# print(\"Number of cpu:\", N_CPU)","metadata":{"execution":{"iopub.status.busy":"2023-01-31T07:23:32.200157Z","iopub.execute_input":"2023-01-31T07:23:32.200957Z","iopub.status.idle":"2023-01-31T07:23:32.342221Z","shell.execute_reply.started":"2023-01-31T07:23:32.200856Z","shell.execute_reply":"2023-01-31T07:23:32.341104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LEAK_DATA = True\nNUM_CHUNK = 10\n\nMIN_TS = 1661724000\nMAX_TS = 1662328791\nprint(\"Starting point of TESTA:\", datetime.datetime.fromtimestamp(MIN_TS))\nprint(\"Ending point of TESTA:\", datetime.datetime.fromtimestamp(MAX_TS))","metadata":{"execution":{"iopub.status.busy":"2023-01-31T07:23:32.344437Z","iopub.execute_input":"2023-01-31T07:23:32.344798Z","iopub.status.idle":"2023-01-31T07:23:32.351436Z","shell.execute_reply.started":"2023-01-31T07:23:32.344766Z","shell.execute_reply":"2023-01-31T07:23:32.350090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type_labels = {'clicks':0, 'carts':1, 'orders':2}\ndef read_file_to_cache(f):\n    df = pd.read_parquet(f)\n    df.ts = (df.ts/1000).astype('int32')\n    df['type'] = df['type'].map(type_labels).astype('int8')\n    return df\n\ntest_files = sorted(glob.glob('/kaggle/input/otto-chunk-data-inparquet-format/test_parquet/*'))\nfiles = [test_files[i] for i in range(len(test_files))]\n\ndfs = [read_file_to_cache(f) for f in files]\ntest_df = pd.concat(dfs, axis=0)\n\ndel dfs\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-31T07:23:35.164006Z","iopub.execute_input":"2023-01-31T07:23:35.164452Z","iopub.status.idle":"2023-01-31T07:23:38.676201Z","shell.execute_reply.started":"2023-01-31T07:23:35.164408Z","shell.execute_reply":"2023-01-31T07:23:38.674828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# features for history candidates\nhis_df = pd.get_dummies(data=test_df, columns=['type'])\nhis_df = his_df.groupby(['session','aid']).agg({'type_0':'sum','type_1':'sum','type_2':'sum','ts':['min','max']})\nhis_df = his_df.reset_index()\nhis_df.columns = ['session','aid','his_num_clicks','his_num_carts','his_num_orders','his_min_ts','his_max_ts']\nhis_df['his_min_ts'] = ((his_df['his_min_ts'] - MIN_TS)/(MAX_TS-MIN_TS)).astype('float32')\nhis_df['his_max_ts'] = ((his_df['his_max_ts'] - MIN_TS)/(MAX_TS-MIN_TS)).astype('float32')\nhis_df['his_num_clicks'] = his_df['his_num_clicks'].astype('int32')\nhis_df['his_num_carts'] = his_df['his_num_carts'].astype('int32')\nhis_df['his_num_orders'] = his_df['his_num_orders'].astype('int32')\nhis_df['his_num_actions'] = his_df['his_num_clicks'] + his_df['his_num_carts'] + his_df['his_num_orders']","metadata":{"execution":{"iopub.status.busy":"2023-01-31T07:23:38.678186Z","iopub.execute_input":"2023-01-31T07:23:38.678580Z","iopub.status.idle":"2023-01-31T07:23:45.310831Z","shell.execute_reply.started":"2023-01-31T07:23:38.678546Z","shell.execute_reply":"2023-01-31T07:23:45.309580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Session features","metadata":{}},{"cell_type":"code","source":"session_features = pd.read_parquet(\"/kaggle/input/otto-timestamp-features-test/sess_feature.parquet\")\nsession_features = session_features[['session', 'ts_max',\n       'ts_min', 'ts_mean']]\nsession_features['ts_max'] = ((session_features['ts_max'] - MIN_TS)/(MAX_TS-MIN_TS)).astype('float32')\nsession_features['ts_min'] = ((session_features['ts_min'] - MIN_TS)/(MAX_TS-MIN_TS)).astype('float32')\nsession_features['ts_mean'] = ((session_features['ts_mean'] - MIN_TS)/(MAX_TS-MIN_TS)).astype('float32')\n\ncolumns = []\nfor f in session_features.columns:\n    columns.append(f if \"sess\" in f else (\"sess_\" + f))\nsession_features.columns = columns\nprint(session_features.columns)\nprint(session_features.shape)","metadata":{"execution":{"iopub.status.busy":"2023-01-31T07:23:45.312224Z","iopub.execute_input":"2023-01-31T07:23:45.312556Z","iopub.status.idle":"2023-01-31T07:23:47.235971Z","shell.execute_reply.started":"2023-01-31T07:23:45.312526Z","shell.execute_reply":"2023-01-31T07:23:47.233991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Item features","metadata":{}},{"cell_type":"code","source":"# aid features from test set\ntest_item_features = pd.get_dummies(data=test_df, columns=['type'])\ntest_item_features = test_item_features.groupby(['aid']).agg({'type_0':'sum','type_1':'sum','type_2':'sum','ts':['min','max']})\ntest_item_features = test_item_features.reset_index()\ntest_item_features.columns = ['aid','aid_test_num_clicks','aid_test_num_carts','aid_test_num_orders','aid_test_min_ts','aid_test_max_ts']\ntest_item_features['aid_test_min_ts'] = ((test_item_features['aid_test_min_ts'] - MIN_TS)/(MAX_TS-MIN_TS)).astype('float32')\ntest_item_features['aid_test_max_ts'] = ((test_item_features['aid_test_max_ts'] - MIN_TS)/(MAX_TS-MIN_TS)).astype('float32')\ntest_item_features['aid_test_num_clicks'] = test_item_features['aid_test_num_clicks'].astype('int32')\ntest_item_features['aid_test_num_carts'] = test_item_features['aid_test_num_carts'].astype('int32')\ntest_item_features['aid_test_num_orders'] = test_item_features['aid_test_num_orders'].astype('int32')\ntest_item_features['aid_test_num_actions'] = test_item_features['aid_test_num_clicks'] + test_item_features['aid_test_num_carts'] + test_item_features['aid_test_num_orders']\nprint(test_item_features.shape)","metadata":{"execution":{"iopub.status.busy":"2023-01-31T07:10:38.731941Z","iopub.execute_input":"2023-01-31T07:10:38.732422Z","iopub.status.idle":"2023-01-31T07:10:40.973122Z","shell.execute_reply.started":"2023-01-31T07:10:38.732378Z","shell.execute_reply":"2023-01-31T07:10:40.971973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# aid features from train + test\nitem_features = pd.read_parquet(\"/kaggle/input/otto-item-features/item_features_full_test.pqt\")\nitem_features.drop(columns=['aid_aid_count'], inplace=True)\nitem_features['aid_ts_min'] = ((item_features['aid_ts_min'] - 1659909600)/(MAX_TS-1659909600)).astype('float32')\nitem_features['aid_ts_max'] = ((item_features['aid_ts_max'] - 1659909600)/(MAX_TS-1659909600)).astype('float32')\nprint(item_features.columns)\nprint(item_features.shape)","metadata":{"execution":{"iopub.status.busy":"2023-01-31T07:25:01.958165Z","iopub.execute_input":"2023-01-31T07:25:01.958557Z","iopub.status.idle":"2023-01-31T07:25:02.840837Z","shell.execute_reply.started":"2023-01-31T07:25:01.958526Z","shell.execute_reply":"2023-01-31T07:25:02.839541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"item_features2 = pd.read_parquet(\"/kaggle/input/otto-timestamp-features-test/aid_features.parquet\")\nitem_features2 = item_features2[['aid','aid_ca_cl_ratio', 'aid_or_cl_ratio',\n       'aid_or_ca_ratio']]\nprint(item_features2.shape)","metadata":{"execution":{"iopub.status.busy":"2023-01-28T16:56:36.484455Z","iopub.execute_input":"2023-01-28T16:56:36.485202Z","iopub.status.idle":"2023-01-28T16:56:38.965241Z","shell.execute_reply.started":"2023-01-28T16:56:36.485148Z","shell.execute_reply":"2023-01-28T16:56:38.963958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"item_features = item_features.merge(item_features2, on='aid', how='left').merge(test_item_features, on='aid', how='left').fillna(0)\ndel item_features2, test_item_features\ngc.collect()\n\nprint(item_features.columns)","metadata":{"execution":{"iopub.status.busy":"2023-01-28T16:56:45.393017Z","iopub.execute_input":"2023-01-28T16:56:45.393460Z","iopub.status.idle":"2023-01-28T16:56:49.272637Z","shell.execute_reply.started":"2023-01-28T16:56:45.393423Z","shell.execute_reply":"2023-01-28T16:56:49.271277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Interaction features","metadata":{}},{"cell_type":"code","source":"%%time\n\nfor _type in ['click','cart','order']:\n    # import history candidates from covi\n    candidates = pd.read_parquet(f\"/kaggle/input/otto-covi-candidates-test/{_type}s_candidates.pqt\")\n\n    history_candidates = candidates.loc[candidates.type_candidate == 1].reset_index(drop=True)\n    history_candidates.rename(columns={'score':'his_covi_score'}, inplace=True)\n    history_candidates = history_candidates.merge(his_df, on=['session','aid'], how='left')\n    \n    common_test_cands = candidates.loc[candidates.type_candidate == 2].reset_index(drop=True)\n    common_test_cands.rename(columns={'score':'his_covi_score'}, inplace=True)\n    del candidates\n    gc.collect()\n\n    # import  potential candidates\n    dfs_list = []\n    files = glob.glob(f\"/kaggle/input/otto-interaction-features-dataset-test/{_type}*\")\n    for file in files: \n        dfs_list.append(pd.read_parquet(file))    \n    potential_candidates = pd.concat(dfs_list, axis=0, ignore_index=True)\n\n    del dfs_list\n    gc.collect()\n    \n    for chunk in range(NUM_CHUNK):\n        print(f\"{_type} | CHUNK {chunk}\")\n\n        sub_pot_cands = potential_candidates.loc[potential_candidates.session % NUM_CHUNK == chunk].reset_index(drop=True)\n        sub_his_cands = history_candidates.loc[history_candidates.session % NUM_CHUNK == chunk].reset_index(drop=True)\n        sub_common_test_cands = common_test_cands.loc[common_test_cands.session % NUM_CHUNK == chunk].reset_index(drop=True)\n        candidates = pd.concat([sub_his_cands, sub_pot_cands.rename(columns={'aid_y':'aid'}), sub_common_test_cands], \n                               ignore_index=True, axis = 0).fillna(0).sort_values(by=['session'], ignore_index=True)\n        \n        del sub_pot_cands, sub_his_cands, sub_common_test_cands\n        gc.collect()\n\n        candidates = candidates.merge(item_features, on=['aid'], how='left').merge(session_features, on=['session'], how='left').fillna(0)\n        candidates.to_parquet(f\"/kaggle/working/{_type}_features_{chunk}.pqt\")\n\n        del candidates\n        gc.collect()\n\n    del history_candidates, potential_candidates, common_test_cands\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-26T03:21:29.171392Z","iopub.execute_input":"2023-01-26T03:21:29.173180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}