{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\n\nThis notebook contains the code for the feature engineering, modeling and predictions. A big part of the code is based on [this notebook by Chris Deotte](https://www.kaggle.com/code/cdeotte/xgboost-baseline-0-680) since at first it didn't occur to me to load the dataset in chunks to avoid the memory issues that the main dataset size was causing.\nFor the model, we use the [XGBoost](https://xgboost.readthedocs.io/en/latest/) library.","metadata":{}},{"cell_type":"markdown","source":"# Setup\n","metadata":{}},{"cell_type":"code","source":"# Import libraries\nimport gc\nimport jo_wilder\n\nimport pandas as pd\nimport numpy as np\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import GroupKFold\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import f1_score","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:44:56.535008Z","iopub.execute_input":"2023-05-15T08:44:56.535473Z","iopub.status.idle":"2023-05-15T08:44:57.368628Z","shell.execute_reply.started":"2023-05-15T08:44:56.535374Z","shell.execute_reply":"2023-05-15T08:44:57.367483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Setup matplotlib\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:44:57.370475Z","iopub.execute_input":"2023-05-15T08:44:57.370902Z","iopub.status.idle":"2023-05-15T08:44:57.378000Z","shell.execute_reply.started":"2023-05-15T08:44:57.370871Z","shell.execute_reply":"2023-05-15T08:44:57.377035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data loading\n","metadata":{}},{"cell_type":"code","source":"# Path to files\ntest_csv_path = \"/kaggle/input/student-performance-and-game-play/test.csv\"\ntrain_csv_path = \"/kaggle/input/student-performance-and-game-play/train.csv\"\ntarget_labels_csv = \"/kaggle/input/student-performance-and-game-play/train_labels.csv\"","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:44:57.379409Z","iopub.execute_input":"2023-05-15T08:44:57.380034Z","iopub.status.idle":"2023-05-15T08:44:57.391876Z","shell.execute_reply.started":"2023-05-15T08:44:57.379991Z","shell.execute_reply":"2023-05-15T08:44:57.390949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load only session_id column\ntmp = pd.read_csv(train_csv_path, usecols=[0])\ntmp = tmp.groupby(\"session_id\")[\"session_id\"].agg(\"count\")","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:44:57.394260Z","iopub.execute_input":"2023-05-15T08:44:57.394917Z","iopub.status.idle":"2023-05-15T08:46:06.708356Z","shell.execute_reply.started":"2023-05-15T08:44:57.394881Z","shell.execute_reply":"2023-05-15T08:46:06.707382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate chunks and skips\npieces = 20\nchunks = int(np.ceil(len(tmp) / pieces))","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:46:06.709790Z","iopub.execute_input":"2023-05-15T08:46:06.710414Z","iopub.status.idle":"2023-05-15T08:46:06.714664Z","shell.execute_reply.started":"2023-05-15T08:46:06.710380Z","shell.execute_reply":"2023-05-15T08:46:06.713784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"reads = []\nskips = [0]\n\nfor k in range(pieces):\n    a = k * chunks\n    b = (k + 1) * chunks\n\n    if b > len(tmp):\n        b = len(tmp)\n\n    r = tmp.iloc[a:b].sum()\n    reads.append(r)\n    skips.append(skips[-1] + r)\n\nprint(f\"pieces: {pieces} of sizes: {reads}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:46:06.716198Z","iopub.execute_input":"2023-05-15T08:46:06.716801Z","iopub.status.idle":"2023-05-15T08:46:06.733360Z","shell.execute_reply.started":"2023-05-15T08:46:06.716767Z","shell.execute_reply":"2023-05-15T08:46:06.732395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(train_csv_path, nrows=reads[0])\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:46:06.734817Z","iopub.execute_input":"2023-05-15T08:46:06.735409Z","iopub.status.idle":"2023-05-15T08:46:10.264989Z","shell.execute_reply.started":"2023-05-15T08:46:06.735375Z","shell.execute_reply":"2023-05-15T08:46:10.263962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_df = pd.read_csv(target_labels_csv)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:46:10.266444Z","iopub.execute_input":"2023-05-15T08:46:10.266771Z","iopub.status.idle":"2023-05-15T08:46:10.640744Z","shell.execute_reply.started":"2023-05-15T08:46:10.266741Z","shell.execute_reply":"2023-05-15T08:46:10.639846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_df[\"session\"] = target_df.session_id.apply(lambda x: int(x.split(\"_\")[0]))","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:46:10.644889Z","iopub.execute_input":"2023-05-15T08:46:10.645853Z","iopub.status.idle":"2023-05-15T08:46:11.043593Z","shell.execute_reply.started":"2023-05-15T08:46:10.645815Z","shell.execute_reply":"2023-05-15T08:46:11.042325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_df[\"q\"] = target_df.session_id.apply(lambda x: int(x.split(\"_\")[-1][1:]))","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:46:11.047387Z","iopub.execute_input":"2023-05-15T08:46:11.047731Z","iopub.status.idle":"2023-05-15T08:46:11.403302Z","shell.execute_reply.started":"2023-05-15T08:46:11.047700Z","shell.execute_reply":"2023-05-15T08:46:11.402057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_df[\"correct\"] = target_df[\"correct\"].astype(\"int8\")\ntarget_df[\"q\"] = target_df[\"q\"].astype(\"int8\")","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:46:11.404968Z","iopub.execute_input":"2023-05-15T08:46:11.405297Z","iopub.status.idle":"2023-05-15T08:46:11.412515Z","shell.execute_reply.started":"2023-05-15T08:46:11.405267Z","shell.execute_reply":"2023-05-15T08:46:11.411297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:46:11.413993Z","iopub.execute_input":"2023-05-15T08:46:11.414336Z","iopub.status.idle":"2023-05-15T08:46:11.431404Z","shell.execute_reply.started":"2023-05-15T08:46:11.414284Z","shell.execute_reply":"2023-05-15T08:46:11.430158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature engineering\n","metadata":{}},{"cell_type":"code","source":"categorical_cols = [\n    \"event_name\",\n    \"fqid\",\n    \"room_fqid\",\n    \"text\",\n    \"text_fqid\",\n]\n\nnumerical_cols = [\n    \"elapsed_time\",\n    \"level\",\n    \"page\",\n    \"room_coor_x\",\n    \"room_coor_y\",\n    \"screen_coor_x\",\n    \"screen_coor_y\",\n    \"hover_duration\",\n]","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:46:11.433137Z","iopub.execute_input":"2023-05-15T08:46:11.433490Z","iopub.status.idle":"2023-05-15T08:46:11.439260Z","shell.execute_reply.started":"2023-05-15T08:46:11.433457Z","shell.execute_reply":"2023-05-15T08:46:11.438113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"event_list = train_df[\"event_name\"].unique().tolist()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:46:11.440697Z","iopub.execute_input":"2023-05-15T08:46:11.441025Z","iopub.status.idle":"2023-05-15T08:46:11.551794Z","shell.execute_reply.started":"2023-05-15T08:46:11.440995Z","shell.execute_reply":"2023-05-15T08:46:11.550755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"name_list = train_df[\"text\"].unique().tolist()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:46:11.552975Z","iopub.execute_input":"2023-05-15T08:46:11.553527Z","iopub.status.idle":"2023-05-15T08:46:11.627614Z","shell.execute_reply.started":"2023-05-15T08:46:11.553495Z","shell.execute_reply":"2023-05-15T08:46:11.626441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fqid_list = train_df[\"fqid\"].unique().tolist()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:46:11.632039Z","iopub.execute_input":"2023-05-15T08:46:11.632398Z","iopub.status.idle":"2023-05-15T08:46:11.716548Z","shell.execute_reply.started":"2023-05-15T08:46:11.632366Z","shell.execute_reply":"2023-05-15T08:46:11.715381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"room_list = train_df[\"room_fqid\"].unique().tolist()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:46:11.717902Z","iopub.execute_input":"2023-05-15T08:46:11.718763Z","iopub.status.idle":"2023-05-15T08:46:11.842774Z","shell.execute_reply.started":"2023-05-15T08:46:11.718725Z","shell.execute_reply":"2023-05-15T08:46:11.841783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"groupby_cols = [\"session_id\", \"level_group\"]","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:46:11.844062Z","iopub.execute_input":"2023-05-15T08:46:11.844597Z","iopub.status.idle":"2023-05-15T08:46:11.849537Z","shell.execute_reply.started":"2023-05-15T08:46:11.844564Z","shell.execute_reply":"2023-05-15T08:46:11.848037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(train_df):\n    dfs = []\n\n    agg_functions = {c: [\"mean\", \"std\", \"sum\", \"max\", \"min\"] for c in numerical_cols}\n\n    for c, funcs in agg_functions.items():\n        tmp = train_df.groupby(groupby_cols)[c].agg(funcs)\n        tmp.columns = [f\"{c}_{agg_name}\" for agg_name in funcs]\n        dfs.append(tmp)\n\n    for c in categorical_cols:\n        tmp = train_df.groupby(groupby_cols)[c].agg(\"nunique\")\n        tmp.name = f\"{tmp.name}_nunique\"\n        dfs.append(tmp)\n\n    for c in event_list:\n        train_df[c] = (train_df[\"event_name\"] == c).astype(np.int8)\n        \n    for c in event_list:\n        tmp = train_df.groupby(groupby_cols).agg({c: \"sum\", \"elapsed_time\": [\"sum\", \"mean\", \"std\", \"max\", \"min\"]})\n        tmp.columns = [f\"{c}_sum\", f\"{c}_elapsed_time_sum\", f\"{c}_elapsed_time_mean\", f\"{c}_elapsed_time_std\", f\"{c}_elapsed_time_max\", f\"{c}_elapsed_time_min\"]\n        dfs.append(tmp)\n\n    for c in room_list:\n        train_df[c] = (train_df[\"room_fqid\"] == c).astype(np.int8)\n\n    for c in room_list:\n        tmp = train_df.groupby(groupby_cols)[c].agg(\"sum\")\n        tmp.name = f\"{tmp.name}_sum\"\n        dfs.append(tmp)\n\n    # Frequency encoding of fqid\n    fqid_counts = train_df['fqid'].value_counts()\n    train_df['fqid_freq_encoded'] = train_df['fqid'].map(fqid_counts)\n\n    tmp = train_df.groupby(groupby_cols)['fqid_freq_encoded'].agg([\"mean\", \"sum\", \"max\", \"min\"])\n    tmp.columns = [f\"fqid_freq_encoded_{agg_name}\" for agg_name in tmp.columns]\n    dfs.append(tmp)\n\n    train_df.drop(columns=['fqid', 'fqid_freq_encoded'], inplace=True)\n\n    # Frequency encoding of text\n    text_counts = train_df['text'].value_counts()\n    train_df['text_freq_encoded'] = train_df['text'].map(text_counts)\n\n    tmp = train_df.groupby(groupby_cols)['text_freq_encoded'].agg([\"mean\", \"sum\", \"max\", \"min\"])\n    tmp.columns = [f\"text_freq_encoded_{agg_name}\" for agg_name in tmp.columns]\n    dfs.append(tmp)\n\n    train_df.drop(columns=['text', 'text_freq_encoded'], inplace=True)\n    \n    # Event frequency\n    event_freq = (\n        train_df.groupby(groupby_cols)[\"event_name\"]\n        .value_counts()\n        .unstack(fill_value=0)\n    )\n    event_freq.columns = [f\"{c}_freq\" for c in event_freq.columns]\n    dfs.append(event_freq)\n\n    # Session duration\n    session_duration = (\n        train_df.groupby(groupby_cols)[\"elapsed_time\"].max()\n        - train_df.groupby(groupby_cols)[\"elapsed_time\"].min()\n    )\n    session_duration.name = \"session_duration\"\n    dfs.append(session_duration)\n\n    # Event duration\n    event_duration = (\n        train_df.groupby(groupby_cols + [\"event_name\"])[\"elapsed_time\"].max()\n        - train_df.groupby(groupby_cols + [\"event_name\"])[\"elapsed_time\"].min()\n    )\n    event_duration = event_duration.unstack(fill_value=0)\n    event_duration.columns = [f\"{c}_duration\" for c in event_duration.columns]\n    dfs.append(event_duration)\n\n    # Event interval\n    train_df[\"event_interval\"] = train_df.groupby(groupby_cols)[\"elapsed_time\"].diff()\n    event_interval = train_df.groupby(groupby_cols)[\"event_interval\"].mean()\n    event_interval.name = \"event_interval\"\n    dfs.append(event_interval)\n\n    # Unique event combinations\n    unique_event_combinations = train_df.groupby(groupby_cols)[\"event_name\"].apply(\n        lambda x: len(set(x))\n    )\n    unique_event_combinations.name = \"unique_event_combinations\"\n    dfs.append(unique_event_combinations)\n\n    df = pd.concat(dfs, axis=1).fillna(-1)\n    df = df.reset_index().set_index(\"session_id\")\n\n    _ = gc.collect()\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:46:11.851667Z","iopub.execute_input":"2023-05-15T08:46:11.852155Z","iopub.status.idle":"2023-05-15T08:46:11.875037Z","shell.execute_reply.started":"2023-05-15T08:46:11.852105Z","shell.execute_reply":"2023-05-15T08:46:11.873989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Process train_df in chunks\nall_chunks = []\nfor k in range(pieces):\n    rows = 0\n    if k > 0:\n        rows = range(1, skips[k] + 1)\n        train_df = pd.read_csv(train_csv_path, skiprows=rows, nrows=reads[k])\n\n    df = feature_engineer(train_df)\n    all_chunks.append(df)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:46:11.876747Z","iopub.execute_input":"2023-05-15T08:46:11.877120Z","iopub.status.idle":"2023-05-15T08:57:33.899562Z","shell.execute_reply.started":"2023-05-15T08:46:11.877084Z","shell.execute_reply":"2023-05-15T08:57:33.898280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Clean memory\ndel train_df\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:57:33.901868Z","iopub.execute_input":"2023-05-15T08:57:33.902214Z","iopub.status.idle":"2023-05-15T08:57:34.049256Z","shell.execute_reply.started":"2023-05-15T08:57:33.902183Z","shell.execute_reply":"2023-05-15T08:57:34.047985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Concatenate all chunks\ndf = pd.concat(all_chunks, axis=0)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:57:34.050940Z","iopub.execute_input":"2023-05-15T08:57:34.051350Z","iopub.status.idle":"2023-05-15T08:57:34.133902Z","shell.execute_reply.started":"2023-05-15T08:57:34.051301Z","shell.execute_reply":"2023-05-15T08:57:34.132600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:57:34.135564Z","iopub.execute_input":"2023-05-15T08:57:34.135925Z","iopub.status.idle":"2023-05-15T08:57:34.143642Z","shell.execute_reply.started":"2023-05-15T08:57:34.135893Z","shell.execute_reply":"2023-05-15T08:57:34.142577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:57:34.145205Z","iopub.execute_input":"2023-05-15T08:57:34.145566Z","iopub.status.idle":"2023-05-15T08:57:34.176497Z","shell.execute_reply.started":"2023-05-15T08:57:34.145531Z","shell.execute_reply":"2023-05-15T08:57:34.175364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:57:34.177964Z","iopub.execute_input":"2023-05-15T08:57:34.178298Z","iopub.status.idle":"2023-05-15T08:57:34.185018Z","shell.execute_reply.started":"2023-05-15T08:57:34.178268Z","shell.execute_reply":"2023-05-15T08:57:34.184228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train model\n","metadata":{}},{"cell_type":"code","source":"features = [c for c in df.columns if c != \"level_group\"]\nusers = df.index.unique()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:57:34.186502Z","iopub.execute_input":"2023-05-15T08:57:34.187378Z","iopub.status.idle":"2023-05-15T08:57:34.201951Z","shell.execute_reply.started":"2023-05-15T08:57:34.187339Z","shell.execute_reply":"2023-05-15T08:57:34.200831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gkf = GroupKFold(n_splits=10)\noof = pd.DataFrame(\n    data=np.zeros((len(users), 18)),\n    index=users,\n)\nmodels = {}","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:57:34.203261Z","iopub.execute_input":"2023-05-15T08:57:34.204407Z","iopub.status.idle":"2023-05-15T08:57:34.215087Z","shell.execute_reply.started":"2023-05-15T08:57:34.204363Z","shell.execute_reply":"2023-05-15T08:57:34.213917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print(f\"Fold {i + 1} => \", end=\"\")\n\n    xgb_params = {\n        \"objective\": \"binary:logistic\",\n        \"eval_metric\": \"logloss\",\n        \"learning_rate\": 0.05,\n        \"max_depth\": 4,\n        \"n_estimators\": 1000,\n        \"early_stopping_rounds\": 50,\n        \"tree_method\": \"hist\",\n        \"subsample\": 0.8,\n        \"colsample_bytree\": 0.4,\n        \"use_label_encoder\": False,\n    }\n\n    for t in range(1, 19):\n        if t <= 3:\n            grp = \"0-4\"\n        elif t <= 13:\n            grp = \"5-12\"\n        elif t <= 22:\n            grp = \"13-22\"\n\n        # Train data\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = target_df.loc[target_df.q == t].set_index(\"session\").loc[train_users]\n\n        # Valid data\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = target_df.loc[target_df.q == t].set_index(\"session\").loc[valid_users]\n\n        # Train model\n        clf = XGBClassifier(**xgb_params)\n        clf.fit(\n            train_x[features].astype(\"float32\"),\n            train_y[\"correct\"],\n            eval_set=[(valid_x[features].astype(\"float32\"), valid_y[\"correct\"])],\n            verbose=0,\n        )\n        print(f\"{t}({clf.best_ntree_limit}), \", end=\"\")\n\n        # Save model and predict valid oof\n        models[f\"{grp}_{t}\"] = clf\n        oof.loc[valid_users, t - 1] = clf.predict_proba(\n            valid_x[features].astype(\"float32\")\n        )[:, 1]\n\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T08:57:34.220700Z","iopub.execute_input":"2023-05-15T08:57:34.221085Z","iopub.status.idle":"2023-05-15T09:16:10.012063Z","shell.execute_reply.started":"2023-05-15T08:57:34.221053Z","shell.execute_reply":"2023-05-15T09:16:10.010985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CV score\n","metadata":{}},{"cell_type":"code","source":"true = oof.copy()\nfor k in range(18):\n    # Get labels for each question\n    tmp = target_df.loc[target_df.q == k + 1].set_index(\"session\").loc[users]\n    true[k] = tmp.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-05-15T09:16:10.013656Z","iopub.execute_input":"2023-05-15T09:16:10.013992Z","iopub.status.idle":"2023-05-15T09:16:10.130993Z","shell.execute_reply.started":"2023-05-15T09:16:10.013962Z","shell.execute_reply":"2023-05-15T09:16:10.129708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = []\nthresholds = []\n\nbest_score = 0\nbest_threshold = 0\n\nfor threshold in np.arange(0.4, 0.81, 0.01):\n    print(f\"{threshold:.02f}, \", end=\"\")\n    preds = (oof.values.reshape((-1)) > threshold).astype(\"int\")\n    m = f1_score(true.values.reshape((-1)), preds, average=\"macro\")\n    scores.append(m)\n    thresholds.append(threshold)\n    if m > best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-05-15T09:16:10.132600Z","iopub.execute_input":"2023-05-15T09:16:10.132937Z","iopub.status.idle":"2023-05-15T09:16:17.892321Z","shell.execute_reply.started":"2023-05-15T09:16:10.132906Z","shell.execute_reply":"2023-05-15T09:16:17.891127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20, 5))\nplt.plot(thresholds, scores, \"-o\", color=\"blue\")\nplt.scatter([best_threshold], [best_score], color=\"blue\", s=300, alpha=1)\nplt.xlabel(\"Threshold\", size=14)\nplt.ylabel(\"Validation F1 Score\", size=14)\nplt.title(\n    f\"Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}\",\n    size=18,\n)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T09:16:17.893963Z","iopub.execute_input":"2023-05-15T09:16:17.894295Z","iopub.status.idle":"2023-05-15T09:16:18.175157Z","shell.execute_reply.started":"2023-05-15T09:16:17.894265Z","shell.execute_reply":"2023-05-15T09:16:18.174012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"When using optimal threshold...\")\nfor k in range(18):\n    # Compute f1 score for each question\n    m = f1_score(\n        true[k].values, (oof[k].values > best_threshold).astype(\"int\"), average=\"macro\"\n    )\n    print(f\"Q{k}: F1 =\", m)\n\n# Compute overall F1 score\nm = f1_score(\n    true.values.reshape((-1)),\n    (oof.values.reshape((-1)) > best_threshold).astype(\"int\"),\n    average=\"macro\",\n)\nprint(\"==> Overall F1 =\", m)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T09:16:18.176702Z","iopub.execute_input":"2023-05-15T09:16:18.177034Z","iopub.status.idle":"2023-05-15T09:16:18.551823Z","shell.execute_reply.started":"2023-05-15T09:16:18.177004Z","shell.execute_reply":"2023-05-15T09:16:18.550660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Infer test data\n","metadata":{}},{"cell_type":"code","source":"# Create environment\nenv = jo_wilder.make_env()\niter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T09:16:18.553377Z","iopub.execute_input":"2023-05-15T09:16:18.553750Z","iopub.status.idle":"2023-05-15T09:16:18.559123Z","shell.execute_reply.started":"2023-05-15T09:16:18.553700Z","shell.execute_reply":"2023-05-15T09:16:18.558112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Clear memory\ndel target_df, df, oof, true\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T09:16:18.560481Z","iopub.execute_input":"2023-05-15T09:16:18.561322Z","iopub.status.idle":"2023-05-15T09:16:18.704991Z","shell.execute_reply.started":"2023-05-15T09:16:18.561267Z","shell.execute_reply":"2023-05-15T09:16:18.703733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {\"0-4\": (1, 4), \"5-12\": (4, 14), \"13-22\": (14, 19)}\n\nfor test, sample_submission in iter_test:\n    # FEATURE ENGINEER TEST DATA\n    df = feature_engineer(test)\n\n    # INFER TEST DATA\n    grp = test.level_group.values[0]\n    a, b = limits[grp]\n    for t in range(a, b):\n        clf = models[f\"{grp}_{t}\"]\n        p = clf.predict_proba(df[features].astype(\"float32\"))[0, 1]\n        mask = sample_submission.session_id.str.contains(f\"q{t}\")\n        sample_submission.loc[mask, \"correct\"] = int(p > best_threshold)\n\n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T09:16:18.706347Z","iopub.execute_input":"2023-05-15T09:16:18.706786Z","iopub.status.idle":"2023-05-15T09:16:21.931004Z","shell.execute_reply.started":"2023-05-15T09:16:18.706753Z","shell.execute_reply":"2023-05-15T09:16:21.929946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission\n","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"submission.csv\")\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T09:16:21.932563Z","iopub.execute_input":"2023-05-15T09:16:21.933205Z","iopub.status.idle":"2023-05-15T09:16:21.945539Z","shell.execute_reply.started":"2023-05-15T09:16:21.933169Z","shell.execute_reply":"2023-05-15T09:16:21.944413Z"},"trusted":true},"execution_count":null,"outputs":[]}]}