{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. | Import Required Libraries","metadata":{}},{"cell_type":"code","source":"import gc\nimport time\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom tqdm import tqdm\nfrom sklearn.metrics import f1_score\nfrom sklearn.model_selection import GroupKFold\n\nimport xgboost as xgb\nimport lightgbm as lgb\nimport catboost as cat","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-13T14:34:14.179906Z","iopub.execute_input":"2023-06-13T14:34:14.180265Z","iopub.status.idle":"2023-06-13T14:34:17.708945Z","shell.execute_reply.started":"2023-06-13T14:34:14.180237Z","shell.execute_reply":"2023-06-13T14:34:17.708056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. | Load Train Data and Labels","metadata":{}},{"cell_type":"code","source":"TRAIN_PATH = '/kaggle/input/predict-student-performance-from-game-play/train.csv'\nTARGET_PATH = '/kaggle/input/predict-student-performance-from-game-play/train_labels.csv'","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:34:17.711176Z","iopub.execute_input":"2023-06-13T14:34:17.712676Z","iopub.status.idle":"2023-06-13T14:34:17.717599Z","shell.execute_reply.started":"2023-06-13T14:34:17.712644Z","shell.execute_reply":"2023-06-13T14:34:17.716592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# READ USER ID ONLY\ntmp = pd.read_csv(TRAIN_PATH, usecols=[0])\ntmp = tmp.groupby('session_id').session_id.agg('count')\n\n# COMPUTE READS AND SKIPS\nPIECES = 10\nCHUNK = int(np.ceil(len(tmp) / PIECES))\n\nreads = []\nskips = [0]\nfor k in range(PIECES):\n    a = k * CHUNK\n    b = (k + 1) * CHUNK\n    if b > len(tmp): b = len(tmp)\n    r = tmp.iloc[a:b].sum()\n    reads.append(r)\n    skips.append(skips[-1] + r)\n\nprint(f'To avoid memory error, we will read train in {PIECES} pieces of sizes:')\nprint(reads)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-06-13T14:34:17.721782Z","iopub.execute_input":"2023-06-13T14:34:17.724079Z","iopub.status.idle":"2023-06-13T14:35:46.835238Z","shell.execute_reply.started":"2023-06-13T14:34:17.724045Z","shell.execute_reply":"2023-06-13T14:35:46.834080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(TRAIN_PATH, nrows=reads[0])\n\nprint('Train size of first piece:', train.shape)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:35:46.837617Z","iopub.execute_input":"2023-06-13T14:35:46.838033Z","iopub.status.idle":"2023-06-13T14:35:56.180547Z","shell.execute_reply.started":"2023-06-13T14:35:46.838005Z","shell.execute_reply":"2023-06-13T14:35:56.179615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = pd.read_csv(TARGET_PATH)\ntargets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]))\ntargets['q'] = targets.session_id.apply(lambda x: int(x.split('_')[-1][1:]))\n\nprint(targets.shape)\ntargets.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:35:56.182387Z","iopub.execute_input":"2023-06-13T14:35:56.183318Z","iopub.status.idle":"2023-06-13T14:35:57.667777Z","shell.execute_reply.started":"2023-06-13T14:35:56.183285Z","shell.execute_reply":"2023-06-13T14:35:57.666764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. | Feature Engineer","metadata":{}},{"cell_type":"code","source":"CATS = ['event_name', 'fqid', 'room_fqid', 'text']\nNUMS = ['elapsed_time', 'level', 'page', 'room_coor_x', 'room_coor_y',\n        'screen_coor_x', 'screen_coor_y', 'hover_duration']\n\n# https://www.kaggle.com/code/kimtaehun/lightgbm-baseline-with-aggregated-log-data\nEVENTS = ['navigate_click', 'person_click', 'cutscene_click', 'object_click',\n          'map_hover', 'notification_click', 'map_click', 'observation_click',\n          'checkpoint']","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:35:57.668910Z","iopub.execute_input":"2023-06-13T14:35:57.669550Z","iopub.status.idle":"2023-06-13T14:35:57.674737Z","shell.execute_reply.started":"2023-06-13T14:35:57.669516Z","shell.execute_reply":"2023-06-13T14:35:57.673966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(train):\n    dfs = []\n    for c in CATS:\n        tmp = train.groupby(['session_id', 'level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id', 'level_group'])[c].agg('mean')\n        tmp.name = tmp.name + '_mean'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id', 'level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    for c in EVENTS:\n        train[c] = (train.event_name == c).astype('int8')\n    for c in EVENTS + ['elapsed_time']:\n        tmp = train.groupby(['session_id', 'level_group'])[c].agg('sum')\n        tmp.name = tmp.name + '_sum'\n        dfs.append(tmp)\n    train = train.drop(EVENTS, axis=1)\n\n    df = pd.concat(dfs, axis=1)\n    df = df.fillna(-1)\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-06-13T14:35:57.675740Z","iopub.execute_input":"2023-06-13T14:35:57.676237Z","iopub.status.idle":"2023-06-13T14:35:57.689542Z","shell.execute_reply.started":"2023-06-13T14:35:57.676209Z","shell.execute_reply":"2023-06-13T14:35:57.688588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# PROCESS TRAIN DATA IN PIECES\nall_pieces = []\nprint(f'Processing train as {PIECES} pieces to avoid memory error... ')\nfor k in range(PIECES):\n    print(k + 1, ', ', end='')\n    SKIPS = 0\n    if k > 0: SKIPS = range(1, skips[k] + 1)\n    train = pd.read_csv(TRAIN_PATH,\n                        nrows=reads[k], skiprows=SKIPS)\n    df = feature_engineer(train)\n    all_pieces.append(df)\n\n# CONCATENATE ALL PIECES\nprint('\\n')\ndel train\ngc.collect()\ndf = pd.concat(all_pieces, axis=0)\nprint('Shape of all train data after feature engineering:', df.shape)\ndf.head()","metadata":{"_kg_hide-output":false,"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-06-13T14:35:57.691043Z","iopub.execute_input":"2023-06-13T14:35:57.691639Z","iopub.status.idle":"2023-06-13T14:44:57.832757Z","shell.execute_reply.started":"2023-06-13T14:35:57.691608Z","shell.execute_reply":"2023-06-13T14:44:57.831767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. | EDA","metadata":{}},{"cell_type":"markdown","source":"1. `Questions` have normal distribution.\n2. Target class `correct` distribution foreach `question id` is **not balance**.\n3. Class `0` (`incorrect`) has **less row counts** than class `1` (`correct`) on all `question id` except `10`, `13`, and `15`.","metadata":{}},{"cell_type":"markdown","source":"## 4.1. | Extract Data","metadata":{}},{"cell_type":"code","source":"%%time\n\ntrain_x_list = []\ntrain_y_list = []\n\n# For each question q_id from 1 to 18\nfor q_id in range(1, 19):\n\n    # Select data with specific level_group\n    grp = '0-4' if q_id <= 3 else '5-12' if q_id <= 13 else '13-22'\n    train_x = df.loc[df.level_group == grp].copy()\n    \n    # Get the target with the same question id\n    targets_filter_id = targets.loc[targets.q == q_id]\n    \n    # Set the target index into session column and match the train_x index\n    train_y = targets_filter_id.set_index('session').loc[train_x.index.values].copy()\n    \n    # Add the q_id column to the dataframes\n    train_x['q_id'] = q_id\n    train_y['q_id'] = q_id\n    \n    train_x_list.append(train_x)\n    train_y_list.append(train_y)\n\n# Dataframe with new q_id column\ntrain_x = pd.concat(train_x_list) \ntrain_y = pd.concat(train_y_list)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-13T14:44:57.836172Z","iopub.execute_input":"2023-06-13T14:44:57.836802Z","iopub.status.idle":"2023-06-13T14:44:58.349017Z","shell.execute_reply.started":"2023-06-13T14:44:57.836769Z","shell.execute_reply":"2023-06-13T14:44:58.348006Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_x.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_y.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4.2. | Question Distribution","metadata":{}},{"cell_type":"code","source":"question_counts = train_y['q_id'].value_counts().reset_index()\nquestion_counts.columns = ['q_id', 'count']\n\nplt.figure(figsize=(20, 6))\nplt.bar(question_counts['q_id'], question_counts['count'])\nplt.xlabel('Question ID')\nplt.ylabel('Row counts')\nplt.title('Number of row counts foreach Question ID')\nplt.show()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-06-13T14:44:58.352988Z","iopub.execute_input":"2023-06-13T14:44:58.353336Z","iopub.status.idle":"2023-06-13T14:44:58.716368Z","shell.execute_reply.started":"2023-06-13T14:44:58.353307Z","shell.execute_reply":"2023-06-13T14:44:58.715597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4.3. | Class Distribution foreach Question ID","metadata":{}},{"cell_type":"code","source":"class_dist = train_y.groupby('q_id')['correct'].value_counts().reset_index(name='count')\n\nplt.figure(figsize=(20, 6))\nax = sns.barplot(data=class_dist, x='q_id', y='count', hue='correct')\nplt.xlabel('Question ID')\nplt.ylabel('Count')\nplt.title('Class distribution foreach Question ID')\n\n# Adding the text labels on each bar\nfor p in ax.patches:\n    ax.text(\n        p.get_x() + p.get_width() / 2., \n        p.get_height(),\n        '{0:.0f}'.format(\n            p.get_height()\n        ), \n        fontsize=12, \n        ha='center', \n        va='bottom'\n    )\n\nplt.show()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-06-13T14:44:58.717432Z","iopub.execute_input":"2023-06-13T14:44:58.718329Z","iopub.status.idle":"2023-06-13T14:44:59.391348Z","shell.execute_reply.started":"2023-06-13T14:44:58.718297Z","shell.execute_reply":"2023-06-13T14:44:59.390479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. | Model Selection Default Parameters","metadata":{}},{"cell_type":"code","source":"FEATURES = [c for c in df.columns if c != 'level_group']\nprint('We will train with', len(FEATURES), 'features')\nALL_USERS = df.index.unique()\nprint('We will train with', len(ALL_USERS), 'users info')\n\ngkf = GroupKFold(n_splits=5)","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:44:59.392292Z","iopub.execute_input":"2023-06-13T14:44:59.392597Z","iopub.status.idle":"2023-06-13T14:44:59.402129Z","shell.execute_reply.started":"2023-06-13T14:44:59.392572Z","shell.execute_reply":"2023-06-13T14:44:59.401157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5.1. Model Params","metadata":{}},{"cell_type":"code","source":"xgb_oof = pd.DataFrame(data=np.zeros((len(ALL_USERS), 18)), index=ALL_USERS)\nxgb_params = {\n    'objective': 'binary:logistic',\n    'eval_metric': 'logloss',\n    'verbosity': 0,\n    'tree_method': 'hist',\n}","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:44:59.403685Z","iopub.execute_input":"2023-06-13T14:44:59.404523Z","iopub.status.idle":"2023-06-13T14:44:59.414168Z","shell.execute_reply.started":"2023-06-13T14:44:59.404492Z","shell.execute_reply":"2023-06-13T14:44:59.413156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb_oof = pd.DataFrame(data=np.zeros((len(ALL_USERS), 18)), index=ALL_USERS)\nlgbm_params = {\n    'objective': 'binary',\n    'metric': ['binary_logloss'],\n    'verbose': -1,\n}\ncallbacks = [lgb.log_evaluation(period=0)]","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:44:59.415806Z","iopub.execute_input":"2023-06-13T14:44:59.416481Z","iopub.status.idle":"2023-06-13T14:44:59.431367Z","shell.execute_reply.started":"2023-06-13T14:44:59.416442Z","shell.execute_reply":"2023-06-13T14:44:59.430152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_oof = pd.DataFrame(data=np.zeros((len(ALL_USERS), 18)), index=ALL_USERS)\ncat_params = {\n    'loss_function': 'Logloss',\n    'eval_metric': 'Logloss',\n    'verbose': 0,\n}","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:44:59.433048Z","iopub.execute_input":"2023-06-13T14:44:59.433716Z","iopub.status.idle":"2023-06-13T14:44:59.442894Z","shell.execute_reply.started":"2023-06-13T14:44:59.433661Z","shell.execute_reply":"2023-06-13T14:44:59.442028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5.2. | Training Model","metadata":{}},{"cell_type":"code","source":"%%time\n\ntotal_iterations = len(list(gkf.split(X=df, groups=df.index)))\nprogress_bar = tqdm(enumerate(gkf.split(X=df, groups=df.index)), total=total_iterations)\n\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    progress_bar.set_description(f\"Iteration {i}/{total_iterations} ({(i / total_iterations) * 100:.2f}%)\")\n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for q_id in range(1, 19):\n        grp = '0-4' if q_id <= 3 else '5-12' if q_id <= 13 else '13-22'\n\n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = targets.loc[targets.q == q_id].set_index('session').loc[train_users]\n\n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = targets.loc[targets.q == q_id].set_index('session').loc[valid_users]\n\n        # ---------------\n        # XGBoost\n        # ---------------\n        xgbm = xgb.XGBClassifier(**xgb_params)\n        xgbm.fit(\n            train_x[FEATURES].astype('float32'), \n            train_y['correct'],\n            eval_set=[(\n                valid_x[FEATURES].astype('float32'), \n                valid_y['correct']\n            )],\n            verbose=0\n        )\n        \n        xgb_oof.loc[valid_users, q_id - 1] = xgbm.predict_proba(valid_x[FEATURES].astype('float32'))[:, 1]\n        \n        # ---------------\n        # LightGBM\n        # ---------------\n        lgbm = lgb.LGBMClassifier(**lgbm_params)\n        lgbm.fit(\n            train_x[FEATURES].astype('float32'), \n            train_y['correct'],\n            eval_set=[(\n                valid_x[FEATURES].astype('float32'), \n                valid_y['correct']\n            )],\n            callbacks=callbacks\n        )\n        \n        lgb_oof.loc[valid_users, q_id - 1] = lgbm.predict_proba(valid_x[FEATURES].astype('float32'))[:, 1]\n        \n        # ---------------\n        # CatBoost\n        # ---------------\n        catm = cat.CatBoostClassifier(**cat_params)\n        catm.fit(\n            train_x[FEATURES].astype('float32'), \n            train_y['correct'],\n            eval_set=[(\n                valid_x[FEATURES].astype('float32'), \n                valid_y['correct']\n            )],\n            verbose=0\n        )\n        \n        cat_oof.loc[valid_users, q_id - 1] = catm.predict_proba(valid_x[FEATURES].astype('float32'))[:, 1]","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-06-13T14:44:59.444516Z","iopub.execute_input":"2023-06-13T14:44:59.445173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 6. | Evaluation Score","metadata":{}},{"cell_type":"code","source":"def calculate_f1_score(oof):\n    f1_scores = []\n    # PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\n    true = oof.copy()\n    for q_id in range(18):\n        # GET TRUE LABELS\n        tmp = targets.loc[targets.q == q_id + 1].set_index('session').loc[ALL_USERS]\n        true[q_id] = tmp.correct.values\n\n    # FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\n    scores = []\n    thresholds = []\n    best_score = 0\n    best_threshold = 0\n\n    for threshold in np.arange(0.4, 0.81, 0.01):\n        preds = (oof.values.reshape((-1)) > threshold).astype('int')\n        m = f1_score(true.values.reshape((-1)), preds, average='macro')\n        scores.append(m)\n        thresholds.append(threshold)\n        if m > best_score:\n            best_score = m\n            best_threshold = threshold\n\n    for q_id in range(18):\n        # COMPUTE F1 SCORE PER QUESTION\n        m = f1_score(true[k].values, (oof[q_id].values > best_threshold).astype('int'), average='macro')\n        f1_scores.append(m)\n        print(f'F1 Score question {q_id+1}: {m:.2f}')\n\n    # COMPUTE F1 SCORE OVERALL\n    overall_f1 = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1)) > best_threshold).astype('int'), average='macro')\n    print('-'*50)\n    print(f'F1 Score Overall: {overall_f1:.2f}')\n    \n    return f1_scores, overall_f1","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 6.1. | XGBoost","metadata":{}},{"cell_type":"code","source":"%%time\nf1_scores_xgb, overall_f1_xgb = calculate_f1_score(xgb_oof)","metadata":{"_kg_hide-input":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 6.2. | LightGBM","metadata":{}},{"cell_type":"code","source":"%%time\nf1_scores_lgb, overall_f1_lgb = calculate_f1_score(lgb_oof)","metadata":{"_kg_hide-input":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 6.3. | CatBoost","metadata":{}},{"cell_type":"code","source":"%%time\nf1_scores_cat, overall_f1_cat = calculate_f1_score(cat_oof)","metadata":{"_kg_hide-input":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 7. | Plot Evaluation and Summary","metadata":{}},{"cell_type":"markdown","source":"1. `CatBoost` has the **highest overall** `F1-Score`.\n\n2. `XGBoost` has **consistent high** `F1-Score` on different question.\n\n3. All models has **low** `F1-Score` on question `2`, `3`, `12`, `13`, and `18`.","metadata":{}},{"cell_type":"code","source":"# Find the highest F1 score for each question ID\nmax_scores = [max(f1_scores_xgb[i], f1_scores_lgb[i], f1_scores_cat[i]) for i in range(len(f1_scores_xgb))]\n\n# Find the model with the highest F1 score for each question ID\nmax_models = np.argmax(np.array([f1_scores_xgb, f1_scores_lgb, f1_scores_cat]), axis=0)\n\n# Plot the F1 scores with dots indicating the highest score for each question ID\nplt.figure(figsize=(20, 6))\n\nx_values = range(18)  # q_id numbers\nplt.plot(x_values, f1_scores_xgb, label='XGBoost')\nplt.plot(x_values, f1_scores_lgb, label='LightGBM')\nplt.plot(x_values, f1_scores_cat, label='CatBoost')\n\n# Add dots for the highest scores\nfor i in range(len(x_values)):\n    max_model = max_models[i]\n    max_score = max_scores[i]\n    plt.scatter(x_values[i], max_score, color=['blue', 'red', 'green'][max_model],\n                label=f'Highest (Model {max_model + 1})' if i == 0 else '')\n\nplt.xlabel('Question ID')\nplt.ylabel('F1 Score')\nplt.title('F1 Score per Question')\nplt.legend()\nplt.show()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute the highest overall F1 score\nhighest_score = max(overall_f1_xgb, overall_f1_lgb, overall_f1_cat)\n\n# Create a list of model names and scores\nmodel_names = ['XGBoost', 'LightGBM', 'CatBoost']\nmodel_scores = [overall_f1_xgb, overall_f1_lgb, overall_f1_cat]\n\n# Plot the highest overall F1 score using a horizontal bar plot\nplt.figure(figsize=(20, 6))\nplt.barh(model_names, model_scores, alpha=0.7)\n\n# Highlight the highest score with a different color\nfor i, score in enumerate(model_scores):\n    if score == highest_score:\n        plt.barh(model_names[i], score, alpha=0.7, color='red')\n        plt.text(score, i, f'{score:.2f}', va='center')\n    else:\n        plt.text(score, i, f'{score:.2f}', va='center')  # Add the score as text on the bar\n\nplt.xlabel('F1 Score')\nplt.title('Highest Overall F1 Score')\nplt.show()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]}]}