{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45533,"databundleVersionId":5748852,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport tensorflow as tf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:42:43.235088Z","iopub.execute_input":"2024-12-08T09:42:43.235435Z","iopub.status.idle":"2024-12-08T09:42:43.240678Z","shell.execute_reply.started":"2024-12-08T09:42:43.235401Z","shell.execute_reply":"2024-12-08T09:42:43.239736Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:42:45.418008Z","iopub.execute_input":"2024-12-08T09:42:45.418402Z","iopub.status.idle":"2024-12-08T09:42:45.437807Z","shell.execute_reply.started":"2024-12-08T09:42:45.418368Z","shell.execute_reply":"2024-12-08T09:42:45.436596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dtypes = {\n    'elapsed_time': np.int32,\n    'event_name': 'category',\n    'name': 'category',\n    'level': np.uint8,\n    'room_coor_x': np.float32,\n    'room_coor_y': np.float32,\n    'screen_coor_x': np.float32,\n    'screen_coor_y': np.float32,\n    'hover_duration': np.float32,\n    'text': 'category',\n    'fqid': 'category',\n    'room_fqid': 'category',\n    'text_fqid': 'category',\n    'fullscreen': 'category',\n    'hq': 'category',\n    'music': 'category',\n    'level_group': 'category'\n}\n\nchunksize = 50000  \nchunks = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes, chunksize=chunksize)\ndf = pd.concat(chunks, axis=0)\nprint(\"Full train dataset shape is {}\".format(df.shape))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:42:47.727379Z","iopub.execute_input":"2024-12-08T09:42:47.727764Z","iopub.status.idle":"2024-12-08T09:45:30.216654Z","shell.execute_reply.started":"2024-12-08T09:42:47.727731Z","shell.execute_reply":"2024-12-08T09:45:30.215570Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\nlabels['session'] = labels.session_id.apply(lambda x: int(x.split('_')[0]) )\nlabels['q'] = labels.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:45:35.954438Z","iopub.execute_input":"2024-12-08T09:45:35.955346Z","iopub.status.idle":"2024-12-08T09:45:36.980440Z","shell.execute_reply.started":"2024-12-08T09:45:35.955294Z","shell.execute_reply":"2024-12-08T09:45:36.978376Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:45:39.167030Z","iopub.execute_input":"2024-12-08T09:45:39.167473Z","iopub.status.idle":"2024-12-08T09:45:39.194735Z","shell.execute_reply.started":"2024-12-08T09:45:39.167437Z","shell.execute_reply":"2024-12-08T09:45:39.193345Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.figure(figsize=(10, 20))\nplt.subplots_adjust(hspace=0.5, wspace=0.5)\nfor n in range(1, 19):\n    ax = plt.subplot(6, 3, n)\n    plot_df = labels.loc[labels.q == n]\n    plot_df = plot_df['correct'].value_counts(normalize=True) * 100  \n    plot_df.plot(ax=ax, kind=\"bar\", color=['#FFA500', '#FFD700']) \n    ax.set_title(f\"Question {n}\")\n    ax.set_xlabel(\"\")\n    ax.set_ylabel(\"Percentage (%)\")  \n    ax.set_xticklabels(ax.get_xticklabels(), rotation=45)\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:45:42.005736Z","iopub.execute_input":"2024-12-08T09:45:42.006145Z","iopub.status.idle":"2024-12-08T09:45:44.975537Z","shell.execute_reply.started":"2024-12-08T09:45:42.006111Z","shell.execute_reply":"2024-12-08T09:45:44.974590Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"CATEGORICAL = ['event_name', 'name','fqid', 'room_fqid', 'text_fqid']\nNUMERICAL = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:45:53.315708Z","iopub.execute_input":"2024-12-08T09:45:53.316142Z","iopub.status.idle":"2024-12-08T09:45:53.321323Z","shell.execute_reply.started":"2024-12-08T09:45:53.316106Z","shell.execute_reply":"2024-12-08T09:45:53.320330Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineer(dataset_df):\n    dfs = []\n    for c in CATEGORICAL:\n        tmp = dataset_df.groupby(['session_id','level_group'],observed=True)[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMERICAL:\n        tmp = dataset_df.groupby(['session_id','level_group'],observed=True)[c].agg('mean')\n        dfs.append(tmp)\n    for c in NUMERICAL:\n        tmp = dataset_df.groupby(['session_id','level_group'],observed=True)[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    dataset_df = pd.concat(dfs,axis=1)\n    dataset_df = dataset_df.fillna(-1)\n    dataset_df = dataset_df.reset_index()\n    dataset_df = dataset_df.set_index('session_id')\n    return dataset_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:45:57.368682Z","iopub.execute_input":"2024-12-08T09:45:57.369405Z","iopub.status.idle":"2024-12-08T09:45:57.379508Z","shell.execute_reply.started":"2024-12-08T09:45:57.369362Z","shell.execute_reply":"2024-12-08T09:45:57.378539Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = feature_engineer(df)\nprint(\"Full prepared dataset shape is {}\".format(df.shape))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:46:01.201772Z","iopub.execute_input":"2024-12-08T09:46:01.202635Z","iopub.status.idle":"2024-12-08T09:46:42.449196Z","shell.execute_reply.started":"2024-12-08T09:46:01.202590Z","shell.execute_reply":"2024-12-08T09:46:42.447639Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:46:47.047834Z","iopub.execute_input":"2024-12-08T09:46:47.048289Z","iopub.status.idle":"2024-12-08T09:46:47.079427Z","shell.execute_reply.started":"2024-12-08T09:46:47.048234Z","shell.execute_reply":"2024-12-08T09:46:47.078313Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.describe().transpose()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:46:49.681822Z","iopub.execute_input":"2024-12-08T09:46:49.682211Z","iopub.status.idle":"2024-12-08T09:46:49.810018Z","shell.execute_reply.started":"2024-12-08T09:46:49.682174Z","shell.execute_reply":"2024-12-08T09:46:49.809179Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def split_dataset(dataset, test_ratio=0.20):\n    USER_LIST = dataset.index.unique()\n    split = int(len(USER_LIST) * (1 - 0.20))\n    return dataset.loc[USER_LIST[:split]], dataset.loc[USER_LIST[split:]]\n\ntrain_x, valid_x = split_dataset(df)\nprint(\"{} examples in training, {} examples in testing.\".format(\n    len(train_x), len(valid_x)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:46:52.785117Z","iopub.execute_input":"2024-12-08T09:46:52.785551Z","iopub.status.idle":"2024-12-08T09:46:52.865198Z","shell.execute_reply.started":"2024-12-08T09:46:52.785512Z","shell.execute_reply":"2024-12-08T09:46:52.864103Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"VALID_USER_LIST = valid_x.index.unique()\nprediction_df = pd.DataFrame(index=valid_x.index, columns=range(1, 19)) \nmodels = {}  \nresults = []  \nevaluation_dict = {}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:59:29.814013Z","iopub.execute_input":"2024-12-08T09:59:29.815196Z","iopub.status.idle":"2024-12-08T09:59:29.832862Z","shell.execute_reply.started":"2024-12-08T09:59:29.815145Z","shell.execute_reply":"2024-12-08T09:59:29.831426Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import xgboost as xgb\nfrom sklearn.metrics import f1_score\nfor q_no in range(1, 19):\n    # Phân nhóm theo `level_group`\n    if q_no <= 3:\n        grp = '0-4'\n    elif q_no <= 13:\n        grp = '5-12'\n    else:\n        grp = '13-22'\n\n    # Chọn dữ liệu cho nhóm\n    train_df = train_x.loc[train_x.level_group == grp].copy()\n    train_users = train_df.index.values\n    valid_df = valid_x.loc[valid_x.level_group == grp].copy()\n    valid_users = valid_df.index.values\n    \n    # Chọn nhãn cho câu hỏi\n    train_labels = labels.loc[labels.q == q_no].set_index('session').loc[train_users]\n    valid_labels = labels.loc[labels.q == q_no].set_index('session').loc[valid_users]\n\n    # Thêm nhãn vào tập dữ liệu\n    train_df['correct'] = train_labels['correct']\n    valid_df['correct'] = valid_labels['correct']\n\n    # Tách đặc trưng và nhãn\n    X_train = train_df.drop(columns=['level_group', 'correct'])\n    y_train = train_df['correct']\n    X_valid = valid_df.drop(columns=['level_group', 'correct'])\n    y_valid = valid_df['correct']\n\n    # Khởi tạo mô hình XGBClassifier\n    clf = xgb.XGBClassifier(\n        objective='binary:logistic',\n        eval_metric='logloss',\n        max_depth=6,\n        learning_rate=0.1,\n        n_estimators=100,\n        use_label_encoder=False,  # Bắt buộc để tránh warning\n        verbosity=1\n    )\n\n    # Huấn luyện mô hình\n    clf.fit(X_train, y_train)\n\n    # Dự đoán\n    y_pred = clf.predict(X_valid)\n    f1 = f1_score(y_valid, y_pred)\n    evaluation_dict[q_no] = f1\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:44:18.217360Z","iopub.execute_input":"2024-12-08T10:44:18.217877Z","iopub.status.idle":"2024-12-08T10:44:32.409459Z","shell.execute_reply.started":"2024-12-08T10:44:18.217838Z","shell.execute_reply":"2024-12-08T10:44:32.408318Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for name, value in evaluation_dict.items():\n  print(f\"question {name}: accuracy {value:.4f}\")\n\nprint(\"\\nAverage accuracy\", sum(evaluation_dict.values())/18)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:44:35.656367Z","iopub.execute_input":"2024-12-08T10:44:35.656780Z","iopub.status.idle":"2024-12-08T10:44:35.665694Z","shell.execute_reply.started":"2024-12-08T10:44:35.656743Z","shell.execute_reply":"2024-12-08T10:44:35.664407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.metrics import f1_score\n\n# Tạo một danh sách các ngưỡng để thử\nthresholds = np.arange(0.4,0.8,0.01)  # Có thể điều chỉnh bước nhảy (0.05) theo nhu cầu\n\nbest_threshold = 0\nbest_f1_score = 0\n\n# Lặp qua từng ngưỡng và tính F1-score\nfor threshold in thresholds:\n    y_pred_prob = clf.predict_proba(X_valid)[:, 1]  # Dự đoán xác suất lớp 1\n    y_pred = (y_pred_prob > threshold).astype(int)  # Áp dụng ngưỡng\n\n    # Tính F1-score\n    f1 = f1_score(y_valid, y_pred)\n\n    # Nếu F1-score cao hơn, cập nhật ngưỡng và giá trị F1-score tốt nhất\n    if f1 > best_f1_score:\n        best_f1_score = f1\n        best_threshold = threshold\n\n# In ra ngưỡng tối ưu và F1-score\nprint(f\"Best threshold: {best_threshold}, Best F1-score: {best_f1_score}\")\n\n# Áp dụng ngưỡng tối ưu cho dự đoán cuối cùng\ny_pred_prob = clf.predict_proba(X_valid)[:, 1]\ny_pred = (y_pred_prob > best_threshold).astype(int)\n\n# Lưu kết quả dự đoán vào danh sách\nfor session_id, pred in zip(valid_users, y_pred):\n    results.append({\n        'session_id': f\"{session_id}_q{q_no}\",\n        'correct': int(pred)\n    })\n\n# Chuyển kết quả thành DataFrame và xuất ra file CSV\noutput_df = pd.DataFrame(results)\noutput_df.to_csv('predicted.csv', index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:49:48.152804Z","iopub.execute_input":"2024-12-08T10:49:48.153219Z","iopub.status.idle":"2024-12-08T10:49:50.353150Z","shell.execute_reply.started":"2024-12-08T10:49:48.153185Z","shell.execute_reply":"2024-12-08T10:49:50.351548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predicted_df = pd.read_csv('predicted.csv')\nprint(predicted_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:50:38.936625Z","iopub.execute_input":"2024-12-08T10:50:38.937085Z","iopub.status.idle":"2024-12-08T10:50:39.280958Z","shell.execute_reply.started":"2024-12-08T10:50:38.937047Z","shell.execute_reply":"2024-12-08T10:50:39.279736Z"}},"outputs":[],"execution_count":null}]}