{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.12"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":82.491496,"end_time":"2025-01-03T15:19:55.593234","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-01-03T15:18:33.101738","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Feature: Physical Health and Fitness + Demographics + Behavior + processed PCIAT-Total to enhance SII value","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom xgboost import XGBClassifier\nimport pandas as pd\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nimport warnings\nwarnings.filterwarnings('ignore')\n\n","metadata":{"papermill":{"duration":2.84283,"end_time":"2025-01-03T15:18:38.21396","exception":false,"start_time":"2025-01-03T15:18:35.37113","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:34.201494Z","iopub.execute_input":"2025-05-07T02:23:34.201972Z","iopub.status.idle":"2025-05-07T02:23:34.209583Z","shell.execute_reply.started":"2025-05-07T02:23:34.201936Z","shell.execute_reply":"2025-05-07T02:23:34.208096Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\ntrain_path = '/kaggle/input/child-mind-institute-problematic-internet-use/train.csv'\ntest_path = '/kaggle/input/child-mind-institute-problematic-internet-use/test.csv'\n\n\nif os.path.exists(train_path) and os.path.exists(test_path):\n\ttrain_df = pd.read_csv(train_path)\n\ttest_df = pd.read_csv(test_path)\nelse:\n\tprint(\"One or both files do not exist.\")\n    \n","metadata":{"papermill":{"duration":0.092637,"end_time":"2025-01-03T15:18:38.318185","exception":false,"start_time":"2025-01-03T15:18:38.225548","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:34.211605Z","iopub.execute_input":"2025-05-07T02:23:34.212500Z","iopub.status.idle":"2025-05-07T02:23:34.299220Z","shell.execute_reply.started":"2025-05-07T02:23:34.212459Z","shell.execute_reply":"2025-05-07T02:23:34.297806Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"3.Data Preprocessing","metadata":{"papermill":{"duration":0.004132,"end_time":"2025-01-03T15:18:38.326893","exception":false,"start_time":"2025-01-03T15:18:38.322761","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"### Process PCIAT_Total score","metadata":{}},{"cell_type":"code","source":"train_cols = set(train_df.columns)\ntest_cols = set(test_df.columns)\ncolumns_not_in_test = sorted(list(train_cols - test_cols))\n\ncolumns_to_exclude = ['PCIAT-PCIAT_Total', 'PCIAT-Season', 'sii']\nquestion_columns = [\n    col for col in columns_not_in_test if col not in columns_to_exclude\n]\n\nquestion_columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:34.302206Z","iopub.execute_input":"2025-05-07T02:23:34.302715Z","iopub.status.idle":"2025-05-07T02:23:34.312380Z","shell.execute_reply.started":"2025-05-07T02:23:34.302670Z","shell.execute_reply":"2025-05-07T02:23:34.310892Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.scatterplot(data=train_df, x='PCIAT-PCIAT_Total', y='sii', alpha=0.6)\n\nplt.title('Biểu đồ phân tán giữa PCIAT-PCIAT_Total và sii', fontsize=14)\nplt.xlabel('PCIAT-PCIAT_Total', fontsize=12)\nplt.ylabel('sii', fontsize=12)\nplt.grid(True, linestyle='--', alpha=0.5)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:34.314356Z","iopub.execute_input":"2025-05-07T02:23:34.314773Z","iopub.status.idle":"2025-05-07T02:23:34.645795Z","shell.execute_reply.started":"2025-05-07T02:23:34.314735Z","shell.execute_reply":"2025-05-07T02:23:34.644546Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Hàm tính toán lại điểm sii theo PCIAT-Total","metadata":{}},{"cell_type":"code","source":"def recalculate_sii(row):\n    if pd.isna(row['PCIAT-PCIAT_Total']):\n        return np.nan\n    max_possible = row['PCIAT-PCIAT_Total'] + row[question_columns].isna().sum() * 5\n    if row['PCIAT-PCIAT_Total'] <= 30 and max_possible <= 30:\n        return 0\n    elif 31 <= row['PCIAT-PCIAT_Total'] <= 49 and max_possible <= 49:\n        return 1\n    elif 50 <= row['PCIAT-PCIAT_Total'] <= 79 and max_possible <= 79:\n        return 2\n    elif row['PCIAT-PCIAT_Total'] >= 80 and max_possible >= 80:\n        return 3\n    return np.nan\n\ntrain_df['recalc_sii'] = train_df.apply(recalculate_sii, axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:34.646919Z","iopub.execute_input":"2025-05-07T02:23:34.647255Z","iopub.status.idle":"2025-05-07T02:23:36.277719Z","shell.execute_reply.started":"2025-05-07T02:23:34.647228Z","shell.execute_reply":"2025-05-07T02:23:36.276210Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Lọc ra các dòng mà giá trị recalc_sii (tính toán lại) khác với sii (giá trị ban đầu có sẵn).\nChỉ giữ lại các dòng mà sii không phải NaN (để tránh so sánh với giá trị rỗng).\nmismatch_rows là một DataFrame chỉ chứa những dòng có sự khác biệt giữa recalc_sii và sii","metadata":{}},{"cell_type":"code","source":"train_df['recalc_sii'].isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:36.279692Z","iopub.execute_input":"2025-05-07T02:23:36.280281Z","iopub.status.idle":"2025-05-07T02:23:36.309452Z","shell.execute_reply.started":"2025-05-07T02:23:36.280228Z","shell.execute_reply":"2025-05-07T02:23:36.307961Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Dưới đây tìm ra các dòng mà dữ liệu sii 'mơ hồ'. Giải thích: <br>\n=> Vì sau khi tính toán lại dựa trên PCIAT-Total, nếu sii ban đầu khác NaN nhưng recalc_sii lại NaN -> chứng tỏ rằng giá trị sii là 'mơ hồ'.","metadata":{}},{"cell_type":"code","source":"mismatch_rows = train_df[\n    (train_df['recalc_sii'] != train_df['sii']) & train_df['sii'].notna()\n]\n\nmismatch_rows[question_columns + ['recalc_sii']].style.map(\n    lambda x: 'background-color: #FFC0CB' if pd.isna(x) else ''\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:36.311174Z","iopub.execute_input":"2025-05-07T02:23:36.311669Z","iopub.status.idle":"2025-05-07T02:23:36.352915Z","shell.execute_reply.started":"2025-05-07T02:23:36.311625Z","shell.execute_reply":"2025-05-07T02:23:36.351233Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['sii'] = train_df['recalc_sii']\ntrain_df = train_df.drop(mismatch_rows.index)\n\ntrain_df[columns_not_in_test + ['recalc_sii']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:36.357363Z","iopub.execute_input":"2025-05-07T02:23:36.357774Z","iopub.status.idle":"2025-05-07T02:23:36.404670Z","shell.execute_reply.started":"2025-05-07T02:23:36.357745Z","shell.execute_reply":"2025-05-07T02:23:36.403120Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"na_total_rows = train_df[train_df['sii'].isna()]\nna_total_rows","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:36.407515Z","iopub.execute_input":"2025-05-07T02:23:36.408104Z","iopub.status.idle":"2025-05-07T02:23:36.445665Z","shell.execute_reply.started":"2025-05-07T02:23:36.408029Z","shell.execute_reply":"2025-05-07T02:23:36.443667Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = train_df.dropna(subset=['PCIAT-PCIAT_Total'])\ntrain_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:36.447042Z","iopub.execute_input":"2025-05-07T02:23:36.448453Z","iopub.status.idle":"2025-05-07T02:23:36.498010Z","shell.execute_reply.started":"2025-05-07T02:23:36.448405Z","shell.execute_reply":"2025-05-07T02:23:36.496504Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Điền các giá trị bị NaN cho từng câu hỏi (1-20). Lấy giá trị mode (giá trị xuất hiện nhiều nhất) trong cột.","metadata":{}},{"cell_type":"code","source":"for column in question_columns:\n    if train_df[column].isna().any():\n        mode_value = train_df[column].mode()[0]\n        train_df[column] = train_df[column].fillna(mode_value)\n\ntrain_df[columns_not_in_test + ['recalc_sii']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:36.499520Z","iopub.execute_input":"2025-05-07T02:23:36.500204Z","iopub.status.idle":"2025-05-07T02:23:36.568117Z","shell.execute_reply.started":"2025-05-07T02:23:36.500159Z","shell.execute_reply":"2025-05-07T02:23:36.566585Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.drop(columns='recalc_sii', inplace=True)\ntrain_df[columns_not_in_test]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:36.569621Z","iopub.execute_input":"2025-05-07T02:23:36.570143Z","iopub.status.idle":"2025-05-07T02:23:36.618496Z","shell.execute_reply.started":"2025-05-07T02:23:36.570095Z","shell.execute_reply":"2025-05-07T02:23:36.616827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = train_df.drop(columns=question_columns, errors='ignore')\ntrain_df = train_df.drop(columns='PCIAT-PCIAT_Total', errors='ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:36.619896Z","iopub.execute_input":"2025-05-07T02:23:36.620449Z","iopub.status.idle":"2025-05-07T02:23:36.631777Z","shell.execute_reply.started":"2025-05-07T02:23:36.620412Z","shell.execute_reply":"2025-05-07T02:23:36.629995Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Data Cleaning\nTrong phần này, chúng ta sẽ giải quyết các giá trị NaN dựa trên 3 đặc trưng ko NaN (Basic_Demos-Enroll_Season, Basic_Demos-Age, Basic_Demos-Sex).\n\nChúng tôi sẽ xem độ tương quan của 3 feature này với các features còn lại, từ đó sẽ quyết định sử dụng đặc trưng nào điền giá trị NaN.","metadata":{}},{"cell_type":"code","source":"train_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:36.633631Z","iopub.execute_input":"2025-05-07T02:23:36.634178Z","iopub.status.idle":"2025-05-07T02:23:36.653912Z","shell.execute_reply.started":"2025-05-07T02:23:36.634130Z","shell.execute_reply":"2025-05-07T02:23:36.652472Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:36.655549Z","iopub.execute_input":"2025-05-07T02:23:36.656162Z","iopub.status.idle":"2025-05-07T02:23:36.687040Z","shell.execute_reply.started":"2025-05-07T02:23:36.656112Z","shell.execute_reply":"2025-05-07T02:23:36.685659Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Basic_Demos-Age","metadata":{}},{"cell_type":"code","source":"train_df['Basic_Demos-Age'].describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:36.688434Z","iopub.execute_input":"2025-05-07T02:23:36.688853Z","iopub.status.idle":"2025-05-07T02:23:36.702040Z","shell.execute_reply.started":"2025-05-07T02:23:36.688819Z","shell.execute_reply":"2025-05-07T02:23:36.700223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Đếm số lượng học sinh \nstudytime_counts = train_df['Basic_Demos-Age'].value_counts().sort_index()\n\n# Vẽ biểu đồ\nplt.figure(figsize=(8, 5))\nbars = plt.bar(studytime_counts.index.astype(str), studytime_counts.values, color='skyblue', edgecolor='black')\n\n# Thêm tiêu đề và nhãn\nplt.title('Biểu đồ phân phối nhóm tuổi người tham gia', fontsize=14)\nplt.xlabel('Nhóm tuổi', fontsize=12)\nplt.ylabel('Số lượng người tham gia', fontsize=12)\n\n# Ghi số lượng lên đầu cột\nfor bar in bars:\n    yval = bar.get_height()\n    plt.text(bar.get_x() + bar.get_width()/2.0, yval + 1, int(yval), ha='center', va='bottom')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:36.704172Z","iopub.execute_input":"2025-05-07T02:23:36.704696Z","iopub.status.idle":"2025-05-07T02:23:37.080809Z","shell.execute_reply.started":"2025-05-07T02:23:36.704647Z","shell.execute_reply":"2025-05-07T02:23:37.079270Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Chúng tôi sẽ chia các nhóm tuổi thành 4 nhóm để dễ dàng phân tích","metadata":{}},{"cell_type":"code","source":"def apply_age_group(df):\n    df['Age Group'] = pd.cut(\n        df['Basic_Demos-Age'],\n        bins=[4, 12, 18, 22],\n        labels=['Children', 'Adolescents', 'Adults'],\n    )\n    return df\n\ntrain_df = apply_age_group(train_df)\ntest_df = apply_age_group(test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:37.081909Z","iopub.execute_input":"2025-05-07T02:23:37.082218Z","iopub.status.idle":"2025-05-07T02:23:37.093427Z","shell.execute_reply.started":"2025-05-07T02:23:37.082193Z","shell.execute_reply":"2025-05-07T02:23:37.091898Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Chúng tôi sẽ encode các nhóm tuổi này dưới đây","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n\nle = LabelEncoder()\ntrain_df['Age_Group_Label'] = le.fit_transform(train_df['Age Group'])\ntest_df['Age_Group_Label'] = le.fit_transform(test_df['Age Group'])\ntrain_df[['Age Group', 'Age_Group_Label']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:37.094693Z","iopub.execute_input":"2025-05-07T02:23:37.095055Z","iopub.status.idle":"2025-05-07T02:23:37.123895Z","shell.execute_reply.started":"2025-05-07T02:23:37.095028Z","shell.execute_reply":"2025-05-07T02:23:37.122129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Lấy các cột cần xem tương quan\nselected_cols = ['CGAS-CGAS_Score', 'Basic_Demos-Age', 'Basic_Demos-Sex']\n\n# Tính ma trận tương quan\ncorr = train_df[selected_cols].corr()\n\n# Vẽ heatmap\nplt.figure(figsize=(6, 4))\nsns.heatmap(corr, annot=True, cmap='coolwarm', fmt=\".2f\")\nplt.title('Correlation between Basic_Demos and SII')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:37.125298Z","iopub.execute_input":"2025-05-07T02:23:37.125599Z","iopub.status.idle":"2025-05-07T02:23:37.471985Z","shell.execute_reply.started":"2025-05-07T02:23:37.125571Z","shell.execute_reply":"2025-05-07T02:23:37.470277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"season_counts = train_df['CGAS-Season'].value_counts(normalize=True)\nseason_counts","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:37.473053Z","iopub.execute_input":"2025-05-07T02:23:37.473565Z","iopub.status.idle":"2025-05-07T02:23:37.488136Z","shell.execute_reply.started":"2025-05-07T02:23:37.473514Z","shell.execute_reply":"2025-05-07T02:23:37.485893Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### CGAS-CGAS_Score","metadata":{}},{"cell_type":"markdown","source":"=> Ở đây, chúng tôi nhận thấy rằng CGAS-CGAS_Score không tương quan với giá trị sii, những chúng tôi sẽ phân tích thêm thông qua các biểu đồ","metadata":{}},{"cell_type":"code","source":"sns.scatterplot(data=train_df, x='sii', y='CGAS-CGAS_Score', palette='viridis')\nplt.title(\"Relationship between Age and CGAS Score\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:37.489821Z","iopub.execute_input":"2025-05-07T02:23:37.490350Z","iopub.status.idle":"2025-05-07T02:23:37.750821Z","shell.execute_reply.started":"2025-05-07T02:23:37.490304Z","shell.execute_reply":"2025-05-07T02:23:37.749331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_rows = train_df['CGAS-CGAS_Score'].isna()\nmissing_rows ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:37.756639Z","iopub.execute_input":"2025-05-07T02:23:37.757057Z","iopub.status.idle":"2025-05-07T02:23:37.767772Z","shell.execute_reply.started":"2025-05-07T02:23:37.757027Z","shell.execute_reply":"2025-05-07T02:23:37.766280Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##### Có thể thấy rằng các điểm dữ liệu đc phân bố rải rác","metadata":{}},{"cell_type":"markdown","source":"## Physical Measures","metadata":{}},{"cell_type":"code","source":"physical_columns = [\n 'Physical-BMI',\n 'Physical-Height',\n 'Physical-Weight',\n 'Physical-Waist_Circumference',\n 'Physical-Diastolic_BP',\n 'Physical-HeartRate',\n 'Physical-Systolic_BP'\n]\n\nwh_cols = [\n    'Physical-BMI', 'Physical-Height',\n    'Physical-Weight', 'Physical-Waist_Circumference'\n]\n\nheart_cols = [\n 'Physical-Diastolic_BP',\n 'Physical-HeartRate',\n 'Physical-Systolic_BP'\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:37.770043Z","iopub.execute_input":"2025-05-07T02:23:37.770500Z","iopub.status.idle":"2025-05-07T02:23:37.788538Z","shell.execute_reply.started":"2025-05-07T02:23:37.770466Z","shell.execute_reply":"2025-05-07T02:23:37.786741Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df[wh_cols].describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:37.790363Z","iopub.execute_input":"2025-05-07T02:23:37.790679Z","iopub.status.idle":"2025-05-07T02:23:37.842630Z","shell.execute_reply.started":"2025-05-07T02:23:37.790654Z","shell.execute_reply":"2025-05-07T02:23:37.841192Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Nhận xét: BMI và Weight có giá trị 0, đây là giá trị không phù hợp -> Chuyển thành giá trị NaN","metadata":{}},{"cell_type":"code","source":"train_df[wh_cols] = train_df[wh_cols].replace(0, np.nan)\ntest_df[wh_cols] = test_df[wh_cols].replace(0, np.nan)\ntrain_df[wh_cols].describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:37.844311Z","iopub.execute_input":"2025-05-07T02:23:37.844744Z","iopub.status.idle":"2025-05-07T02:23:37.875974Z","shell.execute_reply.started":"2025-05-07T02:23:37.844702Z","shell.execute_reply":"2025-05-07T02:23:37.874459Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df[wh_cols].isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:37.877501Z","iopub.execute_input":"2025-05-07T02:23:37.877958Z","iopub.status.idle":"2025-05-07T02:23:37.889487Z","shell.execute_reply.started":"2025-05-07T02:23:37.877920Z","shell.execute_reply":"2025-05-07T02:23:37.887708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Lấy các cột cần xem tương quan\nselected_cols = ['Physical-Height', 'Physical-Weight', 'Basic_Demos-Age', 'Basic_Demos-Sex']\n\n# Tính ma trận tương quan\ncorr = train_df[selected_cols].corr()\n\n# Vẽ heatmap\nplt.figure(figsize=(6, 4))\nsns.heatmap(corr, annot=True, cmap='coolwarm', fmt=\".2f\")\nplt.title('Correlation between Basic_Demos and SII')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:37.891054Z","iopub.execute_input":"2025-05-07T02:23:37.891500Z","iopub.status.idle":"2025-05-07T02:23:38.211980Z","shell.execute_reply.started":"2025-05-07T02:23:37.891452Z","shell.execute_reply":"2025-05-07T02:23:38.210459Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.boxplot(data=train_df, x='Basic_Demos-Age', y='Physical-Height')\nplt.title(\"CGAS Score Distribution by Sex\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:38.213349Z","iopub.execute_input":"2025-05-07T02:23:38.213857Z","iopub.status.idle":"2025-05-07T02:23:38.954529Z","shell.execute_reply.started":"2025-05-07T02:23:38.213728Z","shell.execute_reply":"2025-05-07T02:23:38.952785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.boxplot(data=train_df, x='Basic_Demos-Age', y='Physical-Weight')\nplt.title(\"CGAS Score Distribution by Sex\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:38.956010Z","iopub.execute_input":"2025-05-07T02:23:38.956508Z","iopub.status.idle":"2025-05-07T02:23:39.348472Z","shell.execute_reply.started":"2025-05-07T02:23:38.956464Z","shell.execute_reply":"2025-05-07T02:23:39.347178Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Nhận xét: Tuổi có ảnh hưởng đến chiều cao và cân nặng, do tùy vào theo từng độ tuổi sẽ có mức cân nặng và chiêu cao khác nhau -> Sử dụng Age để fill missing value cho 2 đặc trưng này.\n\nChúng ta sẽ chuyển đổi dữ liệu cho Height, Weight và BMI","metadata":{}},{"cell_type":"code","source":"sns.boxplot(data=train_df, x='Basic_Demos-Sex', y='Physical-Weight')\nplt.title(\"CGAS Score Distribution by Sex\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:39.349825Z","iopub.execute_input":"2025-05-07T02:23:39.350373Z","iopub.status.idle":"2025-05-07T02:23:39.549917Z","shell.execute_reply.started":"2025-05-07T02:23:39.350336Z","shell.execute_reply":"2025-05-07T02:23:39.548255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lbs_to_kg = 0.453592\ninches_to_cm = 2.54\n\ndef process_physical_BMI(df):\n    df['Physical-Weight'] = df['Physical-Weight'] * lbs_to_kg\n    df['Physical-Height'] = df['Physical-Height'] * inches_to_cm\n    df['Physical-Waist_Circumference'] = df['Physical-Waist_Circumference'] * inches_to_cm\n    \n    df['Physical-BMI'] = np.where(\n        df['Physical-Weight'].notna() & df['Physical-Height'].notna(),\n        df['Physical-Weight'] / ((df['Physical-Height'] / 100) ** 2),\n        np.nan\n    )\n    \n    return df\n\ntrain_df = process_physical_BMI(train_df)\ntest_df = process_physical_BMI(test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:39.551528Z","iopub.execute_input":"2025-05-07T02:23:39.552010Z","iopub.status.idle":"2025-05-07T02:23:39.564095Z","shell.execute_reply.started":"2025-05-07T02:23:39.551963Z","shell.execute_reply":"2025-05-07T02:23:39.562496Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def convert_season_to_numeric(df, season_columns):\n    # Định nghĩa mapping thứ tự cho các mùa\n    season_mapping = {\n        'Spring': 0,\n        'Summer': 1,\n        'Fall': 2,\n        'Winter': 3\n    }\n    \n    # Kiểm tra từng cột trong danh sách\n    for col in season_columns:\n        if col in df.columns:\n            # In ra các giá trị trước khi ánh xạ\n            print(f\"Giá trị trước khi ánh xạ trong cột {col}:\")\n            print(df[col].unique())\n            \n            # Áp dụng mapping\n            df[col] = df[col].map(season_mapping)\n            \n            # In ra các giá trị sau khi ánh xạ\n            print(f\"Giá trị sau khi ánh xạ trong cột {col}:\")\n            print(df[col].unique())\n    \n    return df","metadata":{"papermill":{"duration":0.013273,"end_time":"2025-01-03T15:18:38.34421","exception":false,"start_time":"2025-01-03T15:18:38.330937","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:39.565319Z","iopub.execute_input":"2025-05-07T02:23:39.565840Z","iopub.status.idle":"2025-05-07T02:23:39.586242Z","shell.execute_reply.started":"2025-05-07T02:23:39.565805Z","shell.execute_reply":"2025-05-07T02:23:39.584605Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"season_columns = [\n    'Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n    'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n    'PAQ_A-Season', 'PAQ_C-Season',  'SDS-Season', \n    'PreInt_EduHx-Season'\n]\n\n# Áp dụng hàm cho tập train và test\ntest_df = convert_season_to_numeric(test_df, season_columns)\n\n# Kết quả\nprint(\"Test DataFrame sau khi chuyển đổi:\")\nprint(test_df[season_columns].head())\n\ntrain_df = convert_season_to_numeric(train_df, season_columns)\n","metadata":{"papermill":{"duration":0.091306,"end_time":"2025-01-03T15:18:38.439811","exception":false,"start_time":"2025-01-03T15:18:38.348505","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:39.588245Z","iopub.execute_input":"2025-05-07T02:23:39.588772Z","iopub.status.idle":"2025-05-07T02:23:39.691619Z","shell.execute_reply.started":"2025-05-07T02:23:39.588723Z","shell.execute_reply":"2025-05-07T02:23:39.690235Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['Physical-Weight'].isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:39.692930Z","iopub.execute_input":"2025-05-07T02:23:39.693300Z","iopub.status.idle":"2025-05-07T02:23:39.702263Z","shell.execute_reply.started":"2025-05-07T02:23:39.693269Z","shell.execute_reply":"2025-05-07T02:23:39.700710Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Lấy các cột cần xem tương quan\nselected_cols = ['BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season', 'sii']\n\n# Tính ma trận tương quan\ncorr = train_df[selected_cols].corr()\n\n# Vẽ heatmap\nplt.figure(figsize=(6, 4))\nsns.heatmap(corr, annot=True, cmap='coolwarm', fmt=\".2f\")\nplt.title('Correlation between Basic_Demos and SII')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:39.703854Z","iopub.execute_input":"2025-05-07T02:23:39.704410Z","iopub.status.idle":"2025-05-07T02:23:40.086692Z","shell.execute_reply.started":"2025-05-07T02:23:39.704357Z","shell.execute_reply":"2025-05-07T02:23:40.085184Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = train_df.drop(columns=question_columns, errors='ignore')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:40.087892Z","iopub.execute_input":"2025-05-07T02:23:40.088264Z","iopub.status.idle":"2025-05-07T02:23:40.097258Z","shell.execute_reply.started":"2025-05-07T02:23:40.088234Z","shell.execute_reply":"2025-05-07T02:23:40.095854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"seasonCols = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season', 'PCIAT-Season']\n\ntrain_df = train_df.drop(columns=question_columns, errors='ignore')\ntrain_df = train_df.drop(columns=seasonCols, errors='ignore')\ntrain_df = train_df.drop(columns=['PCIAT-PCIAT_Total', 'Age Group', 'Age_Group_Label'], errors='ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:40.098838Z","iopub.execute_input":"2025-05-07T02:23:40.099535Z","iopub.status.idle":"2025-05-07T02:23:40.119270Z","shell.execute_reply.started":"2025-05-07T02:23:40.099476Z","shell.execute_reply":"2025-05-07T02:23:40.117692Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"filtered_features = train_df.drop(columns=['id', 'sii'])  \n\n# Loại bỏ các hàng có giá trị NaN trong y\ntrain_df = train_df.dropna(subset=['sii'])\nX = filtered_features\ny = train_df['sii']\n\n# Define the preprocessing pipeline\nnum_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('scaler', StandardScaler())\n])\n\npreprocessor = ColumnTransformer(transformers=[\n    ('num', num_transformer, filtered_features.columns.tolist())\n])\n\n# Fit and transform X\npreprocessor.fit(X)\nX = pd.DataFrame(preprocessor.transform(X), columns=filtered_features.columns)\n","metadata":{"papermill":{"duration":0.087704,"end_time":"2025-01-03T15:18:38.539967","exception":false,"start_time":"2025-01-03T15:18:38.452263","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:40.120706Z","iopub.execute_input":"2025-05-07T02:23:40.121119Z","iopub.status.idle":"2025-05-07T02:23:40.207589Z","shell.execute_reply.started":"2025-05-07T02:23:40.121045Z","shell.execute_reply":"2025-05-07T02:23:40.206299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_val, y_train, y_val = train_test_split(X,y, test_size=0.2)","metadata":{"papermill":{"duration":0.016528,"end_time":"2025-01-03T15:18:38.569802","exception":false,"start_time":"2025-01-03T15:18:38.553274","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:40.209008Z","iopub.execute_input":"2025-05-07T02:23:40.209440Z","iopub.status.idle":"2025-05-07T02:23:40.218678Z","shell.execute_reply.started":"2025-05-07T02:23:40.209395Z","shell.execute_reply":"2025-05-07T02:23:40.217189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.svm import LinearSVC, SVC\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier, ExtraTreesClassifier, AdaBoostClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom xgboost import XGBClassifier\nfrom sklearn.model_selection import StratifiedKFold, cross_val_score\n\n\n# Random seed\nseed = 2023\n\n# List of models\nmodels = [\n    #LinearSVC(max_iter=10000, random_state=seed),\n    #SVC(random_state=seed),\n    #KNeighborsClassifier(metric='minkowski', p=2),\n    #LogisticRegression(solver='liblinear', max_iter=1000),\n    #DecisionTreeClassifier(random_state=seed),\n    #RandomForestClassifier(random_state=seed),\n    #ExtraTreesClassifier(random_state=seed),\n    #AdaBoostClassifier(random_state=seed),\n    XGBClassifier(eval_metric='logloss', random_state=seed)    \n]\n\n# Function to generate baseline results\ndef generate_baseline_results(models, X, y, metrics='accuracy', cv=5, plot_results=False):\n    # Define k-fold\n    kfold = StratifiedKFold(n_splits=cv, shuffle=True, random_state=42)\n    entries = []\n    \n    # Loop through each model\n    for model in models:\n        model_name = model.__class__.__name__\n        print(f\"Training: {model_name}\")\n        scores = cross_val_score(model, X, y, scoring=metrics, cv=kfold)\n        # Lưu kết quả của tất cả các mô hình vào entries\n        entries.extend([(model_name, fold_idx, score) for fold_idx, score in enumerate(scores)])\n    \n    # Create DataFrame\n    cv_df = pd.DataFrame(entries, columns=['model_name', 'fold_id', 'accuracy_score'])\n    \n    # Optional: Plot results if specified\n    if plot_results:\n        sns.boxplot(x='model_name', y='accuracy_score', data=cv_df, color='lightblue', showmeans=True)\n        plt.title(\"Boxplot of baseline Model Accuracy using 5-fold cross-validation\")\n        plt.xticks(rotation=45)\n        plt.show()\n    \n    # Summary result\n    mean = cv_df.groupby('model_name')['accuracy_score'].mean()\n    std = cv_df.groupby('model_name')['accuracy_score'].std()\n\n    baseline_results = pd.concat([mean, std], axis=1)\n    baseline_results.columns = ['Mean', 'Standard Deviation']\n\n    # Sort results\n    baseline_results.sort_values(by='Mean', ascending=False, inplace=True)\n\n    return baseline_results\n\n# Chạy hàm và hiển thị kết quả\ncv_results = generate_baseline_results(models, X, y, metrics='accuracy', cv=5, plot_results=False)\n\n# In toàn bộ kết quả\nprint(cv_results)\n","metadata":{"papermill":{"duration":76.010792,"end_time":"2025-01-03T15:19:54.593903","exception":false,"start_time":"2025-01-03T15:18:38.583111","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:40.220032Z","iopub.execute_input":"2025-05-07T02:23:40.220506Z","iopub.status.idle":"2025-05-07T02:23:45.066860Z","shell.execute_reply.started":"2025-05-07T02:23:40.220463Z","shell.execute_reply":"2025-05-07T02:23:45.065744Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"8.Evaluate the Model","metadata":{"papermill":{"duration":0.004713,"end_time":"2025-01-03T15:19:54.603908","exception":false,"start_time":"2025-01-03T15:19:54.599195","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Preprocess the test data\nX_test = test_df.drop(columns=['id'])\nX_test = pd.DataFrame(preprocessor.transform(X_test), columns=filtered_features.columns)\n\n\n\n# Use the trained model to make predictions\nbest_model =  XGBClassifier(solver='liblinear', max_iter=1000)\nbest_model.fit(X_train, y_train)\ny_test_pred = best_model.predict(X_test)\n\n# Create a submission DataFrame\nsubmission = pd.DataFrame({\n    'id': test_df['id'],\n    'sii': y_test_pred\n})\n\n# Save the submission DataFrame to a CSV file\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"Submission file created successfully.\") \nprint(submission)","metadata":{"papermill":{"duration":0.21858,"end_time":"2025-01-03T15:19:54.827354","exception":false,"start_time":"2025-01-03T15:19:54.608774","status":"completed"},"tags":[],"vscode":{"languageId":"r"},"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T02:23:45.067849Z","iopub.execute_input":"2025-05-07T02:23:45.068273Z","iopub.status.idle":"2025-05-07T02:23:46.042503Z","shell.execute_reply.started":"2025-05-07T02:23:45.068236Z","shell.execute_reply":"2025-05-07T02:23:46.041288Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"9.Hyperparameter turning","metadata":{"papermill":{"duration":0.004651,"end_time":"2025-01-03T15:19:54.837329","exception":false,"start_time":"2025-01-03T15:19:54.832678","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"10.Make Predictions on Test Set","metadata":{"papermill":{"duration":0.00469,"end_time":"2025-01-03T15:19:54.847048","exception":false,"start_time":"2025-01-03T15:19:54.842358","status":"completed"},"tags":[]}},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.00518,"end_time":"2025-01-03T15:19:54.857106","exception":false,"start_time":"2025-01-03T15:19:54.851926","status":"completed"},"tags":[],"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}