{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Using Polars\n\nI learnt `polars` from last OTTO competition, which helped our team to do fast feature engineering when we faced `cudf`'s GPU memory errors.\n\nIn this competition, although the data size is not as large as OTTO's, it is still convenient to use `polars` with cpu for easy train-inference adaption. Moreover, from my own experiment, the `polars` is faster than `pandas` in Kaggle's notebook even there is only 2-core CPU avalaible, which restricts the full strength of `polars` parallelism.\n\nAnd last, the current Kaggle environment (2023-02-17) has `polars` supported! You can save about 40s from installing it.","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport sys\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nfrom tqdm import tqdm\nfrom sklearn.metrics import f1_score\nfrom sklearn.model_selection import GroupKFold\n\nfrom catboost import CatBoostClassifier, Pool\n\nlevel_groups = [\"0-4\", \"5-12\", \"13-22\"]\nlevel_groups_reverse = {'0-4': 0, '5-12': 1, '13-22': 2}\nlevels = {'0-4': (0, 5), '5-12': (5, 13), '13-22': (13, 23)}\nquestions = {'0-4': (1, 4), '5-12': (4, 14), '13-22': (14, 19)}\nEVENTS = ['checkpoint', 'cutscene_click', 'map_click', 'map_hover', 'navigate_click', 'notebook_click', 'notification_click', 'object_click', 'object_hover', 'observation_click', 'person_click']\nCATS = ['event_name', 'name','fqid', 'room_fqid', 'text_fqid']\nNUMS = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y', 'hover_duration', \"time_past\"]","metadata":{"execution":{"iopub.status.busy":"2023-02-19T15:00:52.113743Z","iopub.execute_input":"2023-02-19T15:00:52.114153Z","iopub.status.idle":"2023-02-19T15:00:52.122514Z","shell.execute_reply.started":"2023-02-19T15:00:52.114120Z","shell.execute_reply":"2023-02-19T15:00:52.121483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = [\n    pl.col(\"page\").cast(pl.Float32),\n    (\n        (pl.col(\"elapsed_time\") - pl.col(\"elapsed_time\").shift(1))\n         .fill_null(0)\n         .clip(0, 1e9)\n         .over([\"session_id\", \"level_group\"])\n         .alias(\"time_past\")\n    ),\n]\naggs = [\n    *[pl.col(c).drop_nulls().n_unique().alias(f\"{c}_unique\") for c in CATS],\n    *[pl.col(c).mean().alias(f\"{c}_mean\") for c in NUMS],\n    *[pl.col(c).std().alias(f\"{c}_std\") for c in NUMS],\n    *[(pl.col(\"event_name\") == c).sum().alias(f\"{c}_sum\") for c in EVENTS],\n]\n\nlabel = (pl.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train_labels.csv\")\n    .select([\n        pl.col(\"session_id\").str.split(\"_q\").arr.get(0).cast(pl.Int64),\n        pl.col(\"session_id\").str.split(\"_q\").arr.get(1).cast(pl.Int32).alias(\"qid\"),\n        pl.col(\"correct\").cast(pl.UInt8)\n    ])\n    .sort([\"session_id\", \"qid\"])\n    .groupby(\"session_id\")\n    .agg(pl.col(\"correct\"))\n    .select([\n        pl.col(\"session_id\"),\n        *[pl.col(\"correct\").arr.get(i).alias(f\"correct_{i + 1}\") for i in range(18)]\n    ])\n)\n\ndf = (pl.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\")\n    .drop([\"fullscreen\", \"hq\", \"music\"])\n    .with_columns(columns)\n    .groupby([\"session_id\", \"level_group\"], maintain_order = True)\n    .agg(aggs)\n    .sort([\"session_id\", \"level_group\"])\n    .join(label, on = \"session_id\", how = \"left\")\n    .fill_null(-1)\n    .to_pandas()\n)\n\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-02-19T14:57:09.322727Z","iopub.execute_input":"2023-02-19T14:57:09.323170Z","iopub.status.idle":"2023-02-19T14:57:46.899863Z","shell.execute_reply.started":"2023-02-19T14:57:09.323137Z","shell.execute_reply":"2023-02-19T14:57:46.898674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = {}\nresults = [[[], []] for _ in range(18)]\nsplit = list(GroupKFold(5).split(df[\"session_id\"].unique(), groups = df[\"session_id\"].unique()))\nfor q in tqdm(range(1, 19)):\n    if q <= 3: grp = '0-4'\n    elif q <= 13: grp = '5-12'\n    elif q <= 22: grp = '13-22'\n    for fold, (train_idx, valid_idx) in enumerate(split):\n        df_train = df.query(\"level_group==@grp\").iloc[train_idx]\n        df_valid = df.query(\"level_group==@grp\").iloc[valid_idx]\n        feature_cols = sorted([_ for _ in df_train.columns if _ not in [\"session_id\", \"level_group\"] and not _.startswith(\"correct_\")])\n        train_pool = Pool(df_train[feature_cols].astype(np.float32), \n                          df_train[f\"correct_{q}\"])\n        valid_pool = Pool(df_valid[feature_cols].astype(np.float32), \n                          df_valid[f\"correct_{q}\"])\n        \n        model = CatBoostClassifier(\n            iterations = 1000,\n            early_stopping_rounds = 50,\n            depth = 4,\n            learning_rate = 0.05,\n            loss_function = \"Logloss\",\n            random_seed = 0,\n            metric_period = 1,\n            subsample = 0.8,\n            colsample_bylevel = 0.4,\n            verbose = 0,\n        )\n        \n        model = model.fit(train_pool, eval_set = valid_pool)\n        \n        y = valid_pool.get_label()\n        yhat = model.predict_proba(valid_pool)[:,1]\n\n        models[(fold, q)] = model\n        \n        results[q - 1][0].append(y)\n        results[q - 1][1].append(yhat)\nresults = [[np.concatenate(_) for _ in _] for _ in results]","metadata":{"execution":{"iopub.status.busy":"2023-02-19T15:09:26.488586Z","iopub.execute_input":"2023-02-19T15:09:26.489006Z","iopub.status.idle":"2023-02-19T15:11:42.579181Z","shell.execute_reply.started":"2023-02-19T15:09:26.488953Z","shell.execute_reply":"2023-02-19T15:11:42.578374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for (fold, q), model in models.items():\n    model.save_model(f\"fold{fold}_q{q}.cbm\")\n\ntrue = pd.DataFrame(np.stack([_[0] for _ in results]).T)\noof = pd.DataFrame(np.stack([_[1] for _ in results]).T)\n\nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.4, 0.81, 0.01):\n    preds = (oof.values.reshape(-1) > threshold).astype('int')\n    m = f1_score(true.values.reshape(-1), preds, average = 'macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m > best_score:\n        best_score = m\n        best_threshold = threshold\n        \nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([best_threshold], [best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()\n\nprint(f'When using optimal threshold = {best_threshold:.2f}...')\nfor k in range(18):\n    m = f1_score(true[k].values, (oof[k].values > best_threshold).astype('int'), average = 'macro')\n    print(f'Q{k}: F1 =',m)\nm = f1_score(true.values.reshape(-1), (oof.values > best_threshold).reshape(-1).astype('int'), average = 'macro')\nprint('==> Overall F1 =', m)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T15:12:43.848871Z","iopub.execute_input":"2023-02-19T15:12:43.849348Z","iopub.status.idle":"2023-02-19T15:12:48.274776Z","shell.execute_reply.started":"2023-02-19T15:12:43.849312Z","shell.execute_reply":"2023-02-19T15:12:48.272835Z"},"trusted":true},"execution_count":null,"outputs":[]}]}