{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Description\n- V16 test | V17 train | LB 0.578\n- V17 test | V18 train | LB 0.578\n- V18 test | V20 train | 100 cands | LB 0.579x15\n- V20 test | V21 train | 100 cands max_depth=3 | LB 0.578x95\n- V21 test | V24 train | 50 cands | LB 0.578\n- V22 test | V22 train | 150 cands | LB\n- V24 test | V39 train | 100 cands + 50features + 20chunks | LB 0.578 (worst)\n- V25 test | V44 train | 100 cands + 50features + 10chunks (CV better than V20 train) | LB 0.579x10\n- V27 test | V45 train | 24 features + 10 chunks | LB 0.579x20\n- V26 test | v34 train | 100 features + 10 chunks | LB 0.578x90\n- V28 test | V51 train | 24 features + 5 chunks | LB 0.579x30 (nhay 5 bac)\n- V30 test = V28 test, with add top common | better V28 nhay 0 bac\n- V31 test | V54 train | remove session have no gt | LB 0.579x40\n- V34 test | V59 train | 37 features + only gt session | LB 0.580\n- V37 test | V61 train | 37 features + 5 chunks | LB\n- V38 test | V63 train | 37 features + 5 chunks + add top common | LB","metadata":{}},{"cell_type":"code","source":"!pip install scikit-learn==1.0.2 --upgrade\nimport sklearn\nprint(sklearn.__version__)\n\n!pip install xgboost pyarrow fastparquet\nUSE_TPU = True","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nfrom tqdm.notebook import tqdm\nimport os, sys, pickle, glob, gc\nfrom collections import Counter\nimport itertools\n\nfrom sklearn.model_selection import GroupKFold\nimport xgboost as xgb","metadata":{"execution":{"iopub.status.busy":"2023-01-26T04:14:19.789406Z","iopub.execute_input":"2023-01-26T04:14:19.789889Z","iopub.status.idle":"2023-01-26T04:14:20.957702Z","shell.execute_reply.started":"2023-01-26T04:14:19.789806Z","shell.execute_reply":"2023-01-26T04:14:20.956666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_top_common(pred, _type):\n    if len(pred) < 20:\n        top = [aid for aid in top_common_test[_type] if aid not in pred]\n        pred = pred + top[:20-len(pred)]\n    return pred\n\ntype_labels = {'clicks':0, 'carts':1, 'orders':2}\ndef load_test(path):    \n    dfs = []\n    for e, chunk_file in sorted(enumerate(glob.glob(path))):\n        chunk = pd.read_parquet(chunk_file)\n        chunk.ts = (chunk.ts/1000).astype('int32')\n        chunk['type'] = chunk['type'].map(type_labels).astype('int8')\n        dfs.append(chunk)\n    return pd.concat(dfs).reset_index(drop=True) #.astype({\"ts\": \"datetime64[ms]\"})\n\ntest_df = load_test('../input/otto-chunk-data-inparquet-format/test_parquet/*')\nprint('Test data has shape',test_df.shape)\n# test.head()\ntop_common_test = dict()\ntop_common_test['click'] = list(test_df.loc[test_df['type']== 0,'aid'].value_counts().index.values[:20]) \ntop_common_test['cart'] = list(test_df.loc[test_df['type']== 1,'aid'].value_counts().index.values[:20])\ntop_common_test['order'] = list(test_df.loc[test_df['type']== 2,'aid'].value_counts().index.values[:20])\ndel test_df\ngc.collect()\n\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\npreds_df_list = []\nfor _type in ['click','cart','order']:\n    print(f\"{_type}\", end=',')\n    \n#     if _type == 'order':\n#         files = sorted(glob.glob(f\"/kaggle/input/fork-of-otto-feature-engineering-test-order/{_type}_features*.pqt\"))\n#     else:\n    files = sorted(glob.glob(f\"/kaggle/input/otto-candidate-features-test/{_type}_features*.pqt\"))\n    preds_df4type_list = []\n    for file in files:\n        print(f\"file:{file.split('/')[-1]}\", end='...')\n        test_candidates = pd.read_parquet(file)\n        FEATURES = test_candidates.columns[2:]\n        print(\"Number of features: \", len(FEATURES))\n        preds = np.zeros(len(test_candidates))\n        for fold in range(5):\n            model = xgb.Booster()\n            model.load_model(f'/kaggle/input/otto-xgb-train/XGB_fold{fold}_{_type}.xgb')\n            if USE_TPU:\n                model.set_param({'predictor': 'cpu_predictor'})\n            else:\n                model.set_param({'predictor': 'gpu_predictor'})\n            dtest = xgb.DMatrix(data=test_candidates[FEATURES])\n            preds += model.predict(dtest)/5\n\n            del model, dtest\n            gc.collect()\n        predictions = test_candidates[['session','aid']].copy()\n        predictions['pred'] = preds\n\n        predictions = predictions.sort_values(['session','pred'], ascending=[True,False]).reset_index(drop=True)\n        predictions['n'] = predictions.groupby('session').aid.cumcount().astype('int32')\n        predictions = predictions.loc[predictions.n<20]\n        sub = predictions.groupby('session').aid.apply(list)\n        sub = sub.to_frame().reset_index()\n        sub.columns = ['session_type','labels']\n        sub['labels'] = sub['labels'].apply(lambda x: add_top_common(x, _type))\n        sub.labels = sub.labels.apply(lambda x: \" \".join(map(str,x)))\n        sub.session_type = sub.session_type.astype('str')+ \"_\" + _type + \"s\"\n        preds_df4type_list.append(sub)\n                             \n        del test_candidates, predictions\n        gc.collect()\n        print(\"Done!\")\n    \n    preds_df4type = pd.concat(preds_df4type_list, ignore_index=True, axis=0)\n    preds_df_list.append(preds_df4type)\n    del preds_df4type_list\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-26T03:53:38.387129Z","iopub.status.idle":"2023-01-26T03:53:38.387507Z","shell.execute_reply.started":"2023-01-26T03:53:38.387311Z","shell.execute_reply":"2023-01-26T03:53:38.387335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df = pd.concat(preds_df_list, ignore_index=True, axis=0)\npred_df.to_csv(\"submission.csv\", index=False)\nprint(pred_df.shape)\npred_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-26T03:53:38.388409Z","iopub.status.idle":"2023-01-26T03:53:38.388770Z","shell.execute_reply.started":"2023-01-26T03:53:38.388581Z","shell.execute_reply":"2023-01-26T03:53:38.388598Z"},"trusted":true},"execution_count":null,"outputs":[]}]}