{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# CPU Catboost baseline","metadata":{}},{"cell_type":"markdown","source":"## Использование polars вместо Pandas\nУ нас есть ограничения на эффективность решения и при этом для создания модели нам нужно преобразовать очень много признаков. Так как в условиях соревнования написано о запрете GPU cuDF мы использовать не можем, хорошим вариантом будет использование библиотеки polars \n[как в этом решении](https://www.kaggle.com/code/carnozhao/cpu-catboost-baseline-using-polars-train)","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport sys\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nfrom tqdm import tqdm\nfrom sklearn.metrics import f1_score\nfrom sklearn.model_selection import GroupKFold\n\nfrom catboost import CatBoostClassifier, Pool","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-25T15:47:39.934757Z","iopub.execute_input":"2023-02-25T15:47:39.935770Z","iopub.status.idle":"2023-02-25T15:47:40.106645Z","shell.execute_reply.started":"2023-02-25T15:47:39.935721Z","shell.execute_reply":"2023-02-25T15:47:40.105236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature engineering (преобразование входных признаков)","metadata":{}},{"cell_type":"code","source":"# Группы уровней, после каждой группы появляются соответствующие вопросы, \n# поэтому имеет смысл обучать отдельную модель для каждой группы уровней\nlevel_groups = [\"0-4\", \"5-12\", \"13-22\"]\nlevel_groups_reverse = {'0-4': 0, '5-12': 1, '13-22': 2}\nlevels = {'0-4': (0, 5), '5-12': (5, 13), '13-22': (13, 23)}\nquestions = {'0-4': (1, 4), '5-12': (4, 14), '13-22': (14, 19)}\n\n# Виды событий, в основном это клики, подробнее тут https://www.kaggle.com/code/nguyenthicamlai/eda-ml-on-game-play-ongoing\nEVENTS = ['checkpoint', 'cutscene_click', 'map_click', 'map_hover', 'navigate_click', 'notebook_click', 'notification_click', 'object_click', 'object_hover', 'observation_click', 'person_click']\n# Колонки входных данных\nCATS = ['event_name', 'name','fqid', 'room_fqid', 'text_fqid']\nNUMS = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y', 'hover_duration', \"time_past\"]","metadata":{"execution":{"iopub.status.busy":"2023-02-25T15:51:13.730621Z","iopub.execute_input":"2023-02-25T15:51:13.731088Z","iopub.status.idle":"2023-02-25T15:51:13.738956Z","shell.execute_reply.started":"2023-02-25T15:51:13.731048Z","shell.execute_reply":"2023-02-25T15:51:13.737968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = [\n    pl.col(\"page\").cast(pl.Float32), # преобразуем page в числовой признак\n    (\n        (pl.col(\"elapsed_time\") - pl.col(\"elapsed_time\").shift(1))\n         .fill_null(0)\n         .clip(0, 1e9) # максимальное значение=10^9\n         .over([\"session_id\", \"level_group\"])\n         .alias(\"time_past\") # создаём признак time_past - время, которое прошло с прошлого события в сессии\n    ),\n]\n\naggs = [\n    # преобразуем категориальные признаки в количество уникальных значеий категориального признака за сессию\n    *[pl.col(c).drop_nulls().n_unique().alias(f\"{c}_unique\") for c in CATS], \n    # считаем среднее и среднеквадратичное отклонение для каждого числового признака за сессию\n    *[pl.col(c).mean().alias(f\"{c}_mean\") for c in NUMS],\n    *[pl.col(c).std().alias(f\"{c}_std\") for c in NUMS],\n    # считаем количество каждого события за сессию\n    *[(pl.col(\"event_name\") == c).sum().alias(f\"{c}_sum\") for c in EVENTS],\n]\n\nlabel = (pl.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train_labels.csv\")\n.select([\n    pl.col(\"session_id\").str.split(\"_q\").arr.get(0).cast(pl.Int64),\n    pl.col(\"session_id\").str.split(\"_q\").arr.get(1).cast(pl.Int32).alias(\"qid\"),\n    pl.col(\"correct\").cast(pl.UInt8)\n])\n.sort([\"session_id\", \"qid\"])\n.groupby(\"session_id\")\n.agg(pl.col(\"correct\"))\n.select([\n    pl.col(\"session_id\"),\n    *[pl.col(\"correct\").arr.get(i).alias(f\"correct_{i + 1}\") for i in range(18)]\n])\n)\n\ndf = (pl.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\")\n    .drop([\"fullscreen\", \"hq\", \"music\"]) # это nan признаки\n    .with_columns(columns)\n    .groupby([\"session_id\", \"level_group\"], maintain_order = True)\n    .agg(aggs)\n    .sort([\"session_id\", \"level_group\"])\n    .join(label, on = \"session_id\", how = \"left\")\n    .fill_null(-1)\n    .to_pandas()\n)\n\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-02-25T16:21:24.948258Z","iopub.execute_input":"2023-02-25T16:21:24.948681Z","iopub.status.idle":"2023-02-25T16:21:24.964168Z","shell.execute_reply.started":"2023-02-25T16:21:24.948644Z","shell.execute_reply":"2023-02-25T16:21:24.963121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-25T16:22:02.975502Z","iopub.execute_input":"2023-02-25T16:22:02.976515Z","iopub.status.idle":"2023-02-25T16:22:03.014156Z","shell.execute_reply.started":"2023-02-25T16:22:02.976476Z","shell.execute_reply":"2023-02-25T16:22:03.013277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-25T16:22:03.154082Z","iopub.execute_input":"2023-02-25T16:22:03.154517Z","iopub.status.idle":"2023-02-25T16:22:03.183137Z","shell.execute_reply.started":"2023-02-25T16:22:03.154480Z","shell.execute_reply":"2023-02-25T16:22:03.181934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = {}\nresults = [[[], []] for _ in range(18)]\nquestion_number2group = ['0-4']*4 + ['5-12']*10 + ['13-22']*9\n# делаем фолды по сессиям, чтобы не разбивать группы уровней с одинаковой сессией\nsplit = list(GroupKFold(5).split(df[\"session_id\"].unique(), groups = df[\"session_id\"].unique()))\n\nfor q in tqdm(range(1, 19)):\n    grp = question_number2group[q]\n    for fold, (train_idx, valid_idx) in enumerate(split):\n        df_train = df.query(\"level_group==@grp\").iloc[train_idx]\n        df_valid = df.query(\"level_group==@grp\").iloc[valid_idx]\n        feature_cols = sorted([_ for _ in df_train.columns if _ not in [\"session_id\", \"level_group\"] and not _.startswith(\"correct_\")])\n        train_pool = Pool(df_train[feature_cols].astype(np.float32), \n                          df_train[f\"correct_{q}\"])\n        valid_pool = Pool(df_valid[feature_cols].astype(np.float32), \n                          df_valid[f\"correct_{q}\"])\n        \n        model = CatBoostClassifier(\n            iterations = 1000,\n            early_stopping_rounds = 50,\n            depth = 4,\n            learning_rate = 0.05,\n            loss_function = \"Logloss\",\n            random_seed = 0,\n            metric_period = 1,\n            subsample = 0.8,\n            colsample_bylevel = 0.4,\n            verbose = 0,\n        )\n        \n        model = model.fit(train_pool, eval_set = valid_pool)\n        \n        y = valid_pool.get_label()\n        yhat = model.predict_proba(valid_pool)[:,1]\n\n        models[(fold, q)] = model\n        \n        results[q - 1][0].append(y)\n        results[q - 1][1].append(yhat)\nresults = [[np.concatenate(_) for _ in _] for _ in results]","metadata":{"execution":{"iopub.status.busy":"2023-02-25T16:07:30.851251Z","iopub.execute_input":"2023-02-25T16:07:30.851685Z","iopub.status.idle":"2023-02-25T16:09:49.058250Z","shell.execute_reply.started":"2023-02-25T16:07:30.851650Z","shell.execute_reply":"2023-02-25T16:09:49.057285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for (fold, q), model in models.items():\n    model.save_model(f\"fold{fold}_q{q}.cbm\")\n\ntrue = pd.DataFrame(np.stack([_[0] for _ in results]).T)\noof = pd.DataFrame(np.stack([_[1] for _ in results]).T)\n\nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.4, 0.81, 0.01):\n    preds = (oof.values.reshape(-1) > threshold).astype('int')\n    m = f1_score(true.values.reshape(-1), preds, average = 'macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m > best_score:\n        best_score = m\n        best_threshold = threshold\n        \nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([best_threshold], [best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()\n\nprint(f'When using optimal threshold = {best_threshold:.2f}...')\nfor k in range(18):\n    m = f1_score(true[k].values, (oof[k].values > best_threshold).astype('int'), average = 'macro')\n    print(f'Q{k}: F1 =',m)\nm = f1_score(true.values.reshape(-1), (oof.values > best_threshold).reshape(-1).astype('int'), average = 'macro')\nprint('==> Overall F1 =', m)","metadata":{"execution":{"iopub.status.busy":"2023-02-25T16:11:50.585138Z","iopub.execute_input":"2023-02-25T16:11:50.585550Z","iopub.status.idle":"2023-02-25T16:11:55.038659Z","shell.execute_reply.started":"2023-02-25T16:11:50.585507Z","shell.execute_reply":"2023-02-25T16:11:55.037739Z"},"trusted":true},"execution_count":null,"outputs":[]}]}