{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-12T08:29:20.436519Z","iopub.execute_input":"2024-11-12T08:29:20.437786Z","iopub.status.idle":"2024-11-12T08:29:24.753788Z","shell.execute_reply.started":"2024-11-12T08:29:20.437724Z","shell.execute_reply":"2024-11-12T08:29:24.752434Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(r\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\ntest = pd.read_csv(r\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")\ndata_dict = pd.read_csv(r\"/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-11-12T08:29:32.974460Z","iopub.execute_input":"2024-11-12T08:29:32.975027Z","iopub.status.idle":"2024-11-12T08:29:33.075266Z","shell.execute_reply.started":"2024-11-12T08:29:32.974980Z","shell.execute_reply":"2024-11-12T08:29:33.073962Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#計算資料缺失程度\nmissing_data = train.isnull().mean() * 100\nprint(missing_data[missing_data > 0].sort_values(ascending=False))","metadata":{"execution":{"iopub.status.busy":"2024-11-12T08:29:46.933460Z","iopub.execute_input":"2024-11-12T08:29:46.933918Z","iopub.status.idle":"2024-11-12T08:29:46.960677Z","shell.execute_reply.started":"2024-11-12T08:29:46.933863Z","shell.execute_reply":"2024-11-12T08:29:46.959254Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 刪除指定的欄位\ncolumns_to_drop = [\n    'PAQ_A-PAQ_A_Total', 'PAQ_A-Season', 'Fitness_Endurance-Time_Mins', \n    'Fitness_Endurance-Time_Sec', 'Fitness_Endurance-Max_Stage', \n    'Physical-Waist_Circumference', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD_Zone', \n    'FGC-FGC_GSND', 'FGC-FGC_GSD', 'Fitness_Endurance-Season', \n    'PAQ_C-Season', 'PAQ_C-PAQ_C_Total'\n]\n\n# 使用 drop 方法刪除這些欄位\ntrain = train.drop(columns=columns_to_drop)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T08:30:02.574500Z","iopub.execute_input":"2024-11-12T08:30:02.575101Z","iopub.status.idle":"2024-11-12T08:30:02.592683Z","shell.execute_reply.started":"2024-11-12T08:30:02.575025Z","shell.execute_reply":"2024-11-12T08:30:02.591057Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display(train.head())\nprint(train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-11-12T08:30:04.721762Z","iopub.execute_input":"2024-11-12T08:30:04.722236Z","iopub.status.idle":"2024-11-12T08:30:04.768326Z","shell.execute_reply.started":"2024-11-12T08:30:04.722193Z","shell.execute_reply":"2024-11-12T08:30:04.766595Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import KNNImputer\nimport pandas as pd\n\n# 將多個季節類別變量映射為數值\nseason_mapping = {'Spring': 1, 'Summer': 2, 'Fall': 3, 'Winter': 4}\ninverse_season_mapping = {1: 'Spring', 2: 'Summer', 3: 'Fall', 4: 'Winter'}\n\n# 將季節類別映射為數值以進行填補\ntrain['Physical-Season_numeric'] = train['Physical-Season'].map(season_mapping)\ntrain['FGC-Season_numeric'] = train['FGC-Season'].map(season_mapping)\ntrain['PreInt_EduHx-Season_numeric'] = train['PreInt_EduHx-Season'].map(season_mapping)\n\n# 選擇需要填補的欄位進行 KNN 填補\ncolumns_to_impute = ['Physical-Season_numeric', 'FGC-Season_numeric', 'PreInt_EduHx-Season_numeric']\nimputer = KNNImputer(n_neighbors=5)\ntrain[columns_to_impute] = imputer.fit_transform(train[columns_to_impute])\n\n# 將填補後的數值欄位轉回為類別變量\ntrain['Physical-Season'] = train['Physical-Season_numeric'].round().map(inverse_season_mapping)\ntrain['FGC-Season'] = train['FGC-Season_numeric'].round().map(inverse_season_mapping)\ntrain['PreInt_EduHx-Season'] = train['PreInt_EduHx-Season_numeric'].round().map(inverse_season_mapping)\n\n# 刪除中間的數值欄位\ntrain.drop(columns=['Physical-Season_numeric', 'FGC-Season_numeric', 'PreInt_EduHx-Season_numeric'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-11-12T08:30:09.054167Z","iopub.execute_input":"2024-11-12T08:30:09.054628Z","iopub.status.idle":"2024-11-12T08:30:10.621258Z","shell.execute_reply.started":"2024-11-12T08:30:09.054587Z","shell.execute_reply":"2024-11-12T08:30:10.620143Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#缺失比率在25%以下的進行MICE差補(上述季節資料除外)\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\nimport pandas as pd\n\n# 選擇要填補的欄位和相關特徵\ncolumns_to_impute = [\n    'Physical-Systolic_BP', 'Physical-Diastolic_BP', 'Physical-HeartRate',\n    'Physical-BMI', 'Physical-Height', 'Physical-Weight',\n    'PreInt_EduHx-computerinternet_hoursday',\n]\nadditional_features = ['Basic_Demos-Age', 'Basic_Demos-Sex']\nall_features = columns_to_impute + additional_features\n\n# 使用 MICE 進行填補\nimputer = IterativeImputer(max_iter=50, random_state=0)\nimputed_data_mice = imputer.fit_transform(train[all_features])\n\n# 將填補後的數據保存回原始資料集\ntrain[all_features] = pd.DataFrame(imputed_data_mice, columns=all_features)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T08:30:28.978800Z","iopub.execute_input":"2024-11-12T08:30:28.979246Z","iopub.status.idle":"2024-11-12T08:30:29.120646Z","shell.execute_reply.started":"2024-11-12T08:30:28.979204Z","shell.execute_reply":"2024-11-12T08:30:29.119176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#其他變量(包含缺失率超過30%的使用KNN差補)\nfrom sklearn.impute import KNNImputer\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\nimport pandas as pd\n\n# 定義季節映射\nseason_mapping = {'Spring': 1, 'Summer': 2, 'Fall': 3, 'Winter': 4}\ninverse_season_mapping = {1: 'Spring', 2: 'Summer', 3: 'Fall', 4: 'Winter'}\n\n# 將季節類別變量映射為數值\nseason_columns = ['BIA-Season', 'CGAS-Season', 'SDS-Season']\n\nfor col in season_columns:\n    train[col + '_numeric'] = train[col].map(season_mapping)\n\n# 使用 KNNImputer 進行季節填補\nseason_numeric_columns = [col + '_numeric' for col in season_columns]\nknn_imputer = KNNImputer(n_neighbors=5)\ntrain[season_numeric_columns] = knn_imputer.fit_transform(train[season_numeric_columns])\n\n# 將數值填補後的季節欄位轉回為原始類別\nfor col in season_columns:\n    train[col] = train[col + '_numeric'].round().map(inverse_season_mapping)\n    train.drop(columns=[col + '_numeric'], inplace=True)  # 刪除中間數值列\n\n# 對數值變數進行填補（使用多重插補法）\nnumerical_columns_to_impute = [\n    'BIA-BIA_ICW', 'BIA-BIA_FMI', 'BIA-BIA_ECW', 'BIA-BIA_DEE', 'BIA-BIA_BMR', 'BIA-BIA_BMI',\n    'BIA-BIA_BMC', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_FFMI', 'BIA-BIA_FFM', 'BIA-BIA_LDM',\n    'BIA-BIA_TBW', 'BIA-BIA_Fat', 'BIA-BIA_SMM', 'BIA-BIA_LST', 'BIA-BIA_Frame_num',\n    'FGC-FGC_SRL', 'FGC-FGC_SRR', 'FGC-FGC_PU', 'FGC-FGC_CU', 'FGC-FGC_TL', 'CGAS-CGAS_Score',\n    'SDS-SDS_Total_T', 'SDS-SDS_Total_Raw', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR_Zone', \n    'FGC-FGC_PU_Zone', 'FGC-FGC_CU_Zone', 'FGC-FGC_TL_Zone'\n]\n\nmice_imputer = IterativeImputer(max_iter=20, random_state=0)\ntrain[numerical_columns_to_impute] = mice_imputer.fit_transform(train[numerical_columns_to_impute])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T08:30:33.428166Z","iopub.execute_input":"2024-11-12T08:30:33.428581Z","iopub.status.idle":"2024-11-12T08:31:01.668375Z","shell.execute_reply.started":"2024-11-12T08:30:33.428541Z","shell.execute_reply":"2024-11-12T08:31:01.666522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 檢視是否還有缺失值\n# 刪除 'sii' 欄位有缺失值的資料列\ntrain = train.dropna(subset=['sii']).reset_index(drop=True)\nmissing_values = train.isnull().sum()\nprint(missing_values[missing_values > 0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T08:31:05.351512Z","iopub.execute_input":"2024-11-12T08:31:05.352792Z","iopub.status.idle":"2024-11-12T08:31:05.374493Z","shell.execute_reply.started":"2024-11-12T08:31:05.352731Z","shell.execute_reply":"2024-11-12T08:31:05.372675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#把season轉換成類別\ntrain_cat_columns = train.select_dtypes(exclude = 'number').columns\n\nfor season in train_cat_columns:\n    train[season] = train[season].fillna(0)\n    train[season] = train[season].replace({'Spring':1, 'Summer':2, 'Fall':3, 'Winter':4})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T08:31:19.090974Z","iopub.execute_input":"2024-11-12T08:31:19.091460Z","iopub.status.idle":"2024-11-12T08:31:19.138858Z","shell.execute_reply.started":"2024-11-12T08:31:19.091419Z","shell.execute_reply":"2024-11-12T08:31:19.137776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#cleaning the PCIAT features\nPCIAT_cols = [val for val in train.columns[train.columns.str.contains('PCIAT')]]\nprint('Number of PCIAT features = ' , len(PCIAT_cols))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T08:47:56.922517Z","iopub.execute_input":"2024-11-12T08:47:56.923000Z","iopub.status.idle":"2024-11-12T08:47:56.929765Z","shell.execute_reply.started":"2024-11-12T08:47:56.922953Z","shell.execute_reply":"2024-11-12T08:47:56.928431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PCIAT_cols.remove('PCIAT-PCIAT_Total')\ntrain = train.drop(columns = PCIAT_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T08:47:58.021151Z","iopub.execute_input":"2024-11-12T08:47:58.021637Z","iopub.status.idle":"2024-11-12T08:47:58.029668Z","shell.execute_reply.started":"2024-11-12T08:47:58.021589Z","shell.execute_reply":"2024-11-12T08:47:58.028203Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.tail(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T08:48:00.550791Z","iopub.execute_input":"2024-11-12T08:48:00.551284Z","iopub.status.idle":"2024-11-12T08:48:00.594237Z","shell.execute_reply.started":"2024-11-12T08:48:00.551240Z","shell.execute_reply":"2024-11-12T08:48:00.593051Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, accuracy_score\nimport pandas as pd\n\n# 假設 train 資料已經讀入\nX = train.drop(columns=['sii', 'id'])  # 刪除 'sii' 和 'id' 欄位\ny = train['sii']  # 將 'sii' 作為目標變數\n\n# 分割訓練集和測試集\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2 ,random_state=100)\n\n# 訓練隨機森林模型\nmodel = RandomForestClassifier(random_state=42)\nmodel.fit(X_train, y_train)\n\n# 對測試集進行預測\ny_pred = model.predict(X_val)\n\n# 評估模型表現\nprint(\"模型準確率:\", accuracy_score(y_val, y_pred))\nprint(\"分類報告:\\n\", classification_report(y_val, y_pred))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T08:52:00.221982Z","iopub.execute_input":"2024-11-12T08:52:00.223129Z","iopub.status.idle":"2024-11-12T08:52:01.045100Z","shell.execute_reply.started":"2024-11-12T08:52:00.223039Z","shell.execute_reply":"2024-11-12T08:52:01.043865Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nimport pandas as pd\n\n# 假設 train 資料已經載入\nX = train.drop(columns=['sii', 'id'])  # 特徵欄位\ny = train['sii']  # 目標變數\n\n# 分割資料\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=100)\n\n# 訓練隨機森林模型\nmodel = RandomForestClassifier(random_state=42)\nmodel.fit(X_train, y_train)\n\n# 獲取特徵重要性\nfeature_importances = model.feature_importances_\nfeature_weights = pd.DataFrame({\n    'feature': X.columns,\n    'importance': feature_importances\n}).sort_values(by='importance', ascending=False)\n\nprint(\"各因子權重（特徵重要性）:\")\nprint(feature_weights)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T08:51:46.839511Z","iopub.execute_input":"2024-11-12T08:51:46.840032Z","iopub.status.idle":"2024-11-12T08:51:47.625741Z","shell.execute_reply.started":"2024-11-12T08:51:46.839983Z","shell.execute_reply":"2024-11-12T08:51:47.624483Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 接著處理測試集\n","metadata":{}},{"cell_type":"code","source":"# 刪除指定的欄位\ncolumns_to_drop = [\n    'PAQ_A-PAQ_A_Total', 'PAQ_A-Season', 'Fitness_Endurance-Time_Mins', \n    'Fitness_Endurance-Time_Sec', 'Fitness_Endurance-Max_Stage', \n    'Physical-Waist_Circumference', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD_Zone', \n    'FGC-FGC_GSND', 'FGC-FGC_GSD', 'Fitness_Endurance-Season', \n    'PAQ_C-Season', 'PAQ_C-PAQ_C_Total'\n]\n\n# 使用 drop 方法刪除這些欄位\ntest = test.drop(columns=columns_to_drop)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T08:48:11.898756Z","iopub.execute_input":"2024-11-12T08:48:11.899281Z","iopub.status.idle":"2024-11-12T08:48:12.859129Z","shell.execute_reply.started":"2024-11-12T08:48:11.899232Z","shell.execute_reply":"2024-11-12T08:48:12.857422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import KNNImputer\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\nimport pandas as pd\n\n# 假設 test 資料已經讀入\n\n# 定義季節映射\nseason_mapping = {'Spring': 1, 'Summer': 2, 'Fall': 3, 'Winter': 4}\ninverse_season_mapping = {1: 'Spring', 2: 'Summer', 3: 'Fall', 4: 'Winter'}\n\n# 1. 將季節類別變量映射為數值\nseason_columns = ['Physical-Season', 'FGC-Season', 'PreInt_EduHx-Season', 'BIA-Season', 'CGAS-Season', 'SDS-Season']\n\nfor col in season_columns:\n    test[col + '_numeric'] = test[col].map(season_mapping)\n\n# 2. 使用 KNNImputer 填補季節數值欄位\nseason_numeric_columns = [col + '_numeric' for col in season_columns]\nknn_imputer = KNNImputer(n_neighbors=5)\ntest[season_numeric_columns] = knn_imputer.fit_transform(test[season_numeric_columns])\n\n# 3. 將數值填補後的季節欄位轉回為原始類別\nfor col in season_columns:\n    test[col] = test[col + '_numeric'].round().map(inverse_season_mapping)\n    test.drop(columns=[col + '_numeric'], inplace=True)  # 刪除中間數值欄位\n\n# 4. 使用 MICE 進行多重插補法填補數值欄位\n# 定義要填補的數值欄位和相關特徵\nnumerical_columns_to_impute = [\n    'Physical-Systolic_BP', 'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-BMI', \n    'Physical-Height', 'Physical-Weight', 'PreInt_EduHx-computerinternet_hoursday',\n    'BIA-BIA_ICW', 'BIA-BIA_FMI', 'BIA-BIA_ECW', 'BIA-BIA_DEE', 'BIA-BIA_BMR', 'BIA-BIA_BMI',\n    'BIA-BIA_BMC', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_FFMI', 'BIA-BIA_FFM', 'BIA-BIA_LDM',\n    'BIA-BIA_TBW', 'BIA-BIA_Fat', 'BIA-BIA_SMM', 'BIA-BIA_LST', 'BIA-BIA_Frame_num',\n    'FGC-FGC_SRL', 'FGC-FGC_SRR', 'FGC-FGC_PU', 'FGC-FGC_CU', 'FGC-FGC_TL', 'CGAS-CGAS_Score',\n    'SDS-SDS_Total_T', 'SDS-SDS_Total_Raw', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR_Zone', \n    'FGC-FGC_PU_Zone', 'FGC-FGC_CU_Zone', 'FGC-FGC_TL_Zone'\n]\nadditional_features = ['Basic_Demos-Age', 'Basic_Demos-Sex']\nall_features = numerical_columns_to_impute + additional_features\n\n# 使用 MICE 進行填補\nmice_imputer = IterativeImputer(max_iter=100, random_state=0)\ntest[all_features] = mice_imputer.fit_transform(test[all_features])\n\n# 最終填補完成的 test 資料\nprint(\"填補完成的 test 資料:\")\nprint(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T08:34:09.878429Z","iopub.execute_input":"2024-11-12T08:34:09.878899Z","iopub.status.idle":"2024-11-12T08:34:10.077674Z","shell.execute_reply.started":"2024-11-12T08:34:09.878855Z","shell.execute_reply":"2024-11-12T08:34:10.076118Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 檢視是否還有缺失值\n# 刪除 'sii' 欄位有缺失值的資料列\nmissing_values = test.isnull().sum()\nprint(missing_values[missing_values > 0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T08:34:20.302260Z","iopub.execute_input":"2024-11-12T08:34:20.302767Z","iopub.status.idle":"2024-11-12T08:34:20.313350Z","shell.execute_reply.started":"2024-11-12T08:34:20.302723Z","shell.execute_reply":"2024-11-12T08:34:20.312133Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 去除空格，確保欄位名稱正確\ncolumn_name = 'PCIAT-PCIAT_Total'\n\n# 檢查 train 是否包含該欄位，且 test 沒有該欄位\nif column_name in train.columns and column_name not in test.columns:\n    test[column_name] = train[column_name].mean()  # 使用 train 的均值填補\n\n# 查看填補後的值\nprint(test[column_name])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T08:48:49.677175Z","iopub.execute_input":"2024-11-12T08:48:49.677606Z","iopub.status.idle":"2024-11-12T08:48:49.685857Z","shell.execute_reply.started":"2024-11-12T08:48:49.677551Z","shell.execute_reply":"2024-11-12T08:48:49.684297Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T08:48:52.124528Z","iopub.execute_input":"2024-11-12T08:48:52.125119Z","iopub.status.idle":"2024-11-12T08:48:52.169291Z","shell.execute_reply.started":"2024-11-12T08:48:52.125039Z","shell.execute_reply":"2024-11-12T08:48:52.167866Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#把season轉換成類別\ntest_cat_columns = test.select_dtypes(exclude = 'number').columns\n\nfor season in test_cat_columns:\n    test[season] = test[season].fillna(0)\n    test[season] = test[season].replace({'Spring':1, 'Summer':2, 'Fall':3, 'Winter':4})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T08:48:54.547692Z","iopub.execute_input":"2024-11-12T08:48:54.548177Z","iopub.status.idle":"2024-11-12T08:48:54.570990Z","shell.execute_reply.started":"2024-11-12T08:48:54.548131Z","shell.execute_reply":"2024-11-12T08:48:54.569408Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 確保使用與 train 一致的特徵欄位\nX_test = test[X.columns]  # 使用與訓練中相同的特徵欄位\n\n# 使用訓練好的模型對 test 資料集進行預測\ny_test_pred = model.predict(X_test)\n\n# 將預測結果存入 test 資料集\ntest['sii'] = y_test_pred\n\n# 創建提交檔案\nsubmission = test[['id', 'sii']]\nsubmission.to_csv('/kaggle/working/submission.csv', index=False)\n\nprint(submission)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T08:54:09.560468Z","iopub.execute_input":"2024-11-12T08:54:09.560954Z","iopub.status.idle":"2024-11-12T08:54:09.593770Z","shell.execute_reply.started":"2024-11-12T08:54:09.560915Z","shell.execute_reply":"2024-11-12T08:54:09.592405Z"}},"outputs":[],"execution_count":null}]}