{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport re\nimport matplotlib.pyplot as plt\n\n\ntrain = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")\n\ntrain = train.dropna(subset=['PCIAT-PCIAT_Total'])\n\ntrain_col_to_drop= [\n    col for col in train.columns\n    if 'season' in col.lower() or re.match(r'^PCIAT-PCIAT_\\d{2}$', col)\n]\n\ntest_col_to_drop= [\n    col for col in test.columns\n    if 'season' in col.lower() or re.match(r'^PCIAT-PCIAT_\\d{2}$', col)\n]\n\ntrain = train.drop(columns = train_col_to_drop)\ntest = test.drop(columns = test_col_to_drop)\n\n\n#print(train.dtypes)\n#train\n\n#print(test.dtypes)\n#test","metadata":{"execution":{"iopub.status.busy":"2024-11-20T09:09:00.461149Z","iopub.execute_input":"2024-11-20T09:09:00.461539Z","iopub.status.idle":"2024-11-20T09:09:00.945177Z","shell.execute_reply.started":"2024-11-20T09:09:00.461503Z","shell.execute_reply":"2024-11-20T09:09:00.944058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import KNNImputer, SimpleImputer\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\n\n\nx = train.drop(columns=['id', 'PCIAT-PCIAT_Total', 'sii'])\ny = train['PCIAT-PCIAT_Total']\n\n# 確認類別型與數值型欄位\ncategorical_cols = x.select_dtypes(include=['object']).columns\nnumerical_cols = x.select_dtypes(exclude=['object']).columns\n\n# 數值欄位使用 KNN 插補和標準化\nnumerical_transformer = Pipeline(steps=[\n    ('imputer', KNNImputer(n_neighbors=5)),\n    ('scaler', StandardScaler())\n])\n\n# 類別欄位使用眾數填補和 One-Hot 編碼\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))\n])\n\n# 建立前處理器\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, numerical_cols),\n        ('cat', categorical_transformer, categorical_cols)\n    ])\n\n# 應用前處理到數據\nx_processed = preprocessor.fit_transform(x)\n\nx_train, x_test, y_train, y_test = train_test_split(x_processed, y, test_size=0.2, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-20T09:09:00.947113Z","iopub.execute_input":"2024-11-20T09:09:00.947514Z","iopub.status.idle":"2024-11-20T09:09:04.260124Z","shell.execute_reply.started":"2024-11-20T09:09:00.947478Z","shell.execute_reply":"2024-11-20T09:09:04.259242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_absolute_error\nfrom lightgbm import LGBMRegressor, early_stopping, log_evaluation\n\n\n# 初始化 LightGBM\nlgbm = LGBMRegressor(objective='regression_l1', random_state=42)\n\nparams = {\n    'larning rate': 0.1,\n}\n\n# 訓練模型，啟用 early stopping\nlgbm.fit(\n    x_train, y_train,\n    eval_set=[(x_train, y_train), (x_test, y_test)],\n    eval_metric='l1',\n    callbacks=[early_stopping(stopping_rounds=10), log_evaluation(0)]\n)\n\n# 獲取評估歷史\nevals_result = lgbm.evals_result_\n\n# 繪製損失曲線\nplt.figure(figsize=(10, 6))\nplt.plot(evals_result['training']['l1'], label='Train Loss')\nplt.plot(evals_result['valid_1']['l1'], label='Validation MAE')\nplt.title('LightGBM Training and Validation Loss')\nplt.xlabel('Iterations')\nplt.ylabel('MAE')\nplt.legend()\nplt.grid()\nplt.show()\n\n\n# 預測與評估\ny_pred_lgbm = lgbm.predict(x_test)\nmae_lgbm = mean_absolute_error(y_test, y_pred_lgbm)\nprint(f\"LightGBM MAE: {mae_lgbm:.4f}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-20T09:09:04.261503Z","iopub.execute_input":"2024-11-20T09:09:04.261976Z","iopub.status.idle":"2024-11-20T09:09:05.848121Z","shell.execute_reply.started":"2024-11-20T09:09:04.261940Z","shell.execute_reply":"2024-11-20T09:09:05.846961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from xgboost import XGBRegressor\n\n# 初始化 XGBoost\nxgb = XGBRegressor(\n    objective='reg:squarederror',\n    eval_metric='mae',\n    early_stopping_rounds=10,\n    random_state=42\n)\n\n\n# 訓練模型\nxgb.fit(\n    x_train, y_train,\n    eval_set=[(x_train, y_train), (x_test, y_test)],\n    verbose=False\n)\n\n# 獲取評估歷史\nevals_result = xgb.evals_result()\n\n\n# 繪製損失曲線\nplt.figure(figsize=(10, 6))\nplt.plot(evals_result['validation_0']['mae'], label='Train Loss')\nplt.plot(evals_result['validation_1']['mae'], label='Validation MAE')\nplt.title('XGBoost Training and Validation Loss')\nplt.xlabel('Iterations')\nplt.ylabel('MAE')\nplt.legend()\nplt.grid()\nplt.show()\n\n\n# 預測與評估\ny_pred_xgb = xgb.predict(x_test)\nmae_xgb = mean_absolute_error(y_test, y_pred_xgb)\nprint(f\"XGBoost MAE: {mae_xgb:.4f}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-20T09:09:05.849630Z","iopub.execute_input":"2024-11-20T09:09:05.850399Z","iopub.status.idle":"2024-11-20T09:09:06.532670Z","shell.execute_reply.started":"2024-11-20T09:09:05.850344Z","shell.execute_reply":"2024-11-20T09:09:06.531523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from catboost import CatBoostRegressor\n\n# 初始化 CatBoost\ncat = CatBoostRegressor(loss_function='MAE', random_seed=42, verbose=0)\n\n# 訓練模型，啟用 early stopping\ncat.fit(\n    x_train, y_train,\n    eval_set=(x_test, y_test),\n    early_stopping_rounds=10\n)\n\n# 獲取評估歷史\nevals_result = cat.get_evals_result()\n\n\n# 繪製損失曲線\nplt.figure(figsize=(10, 6))\nplt.plot(evals_result['learn']['MAE'], label='Train Loss')\nplt.plot(evals_result['validation']['MAE'], label='Validation MAE')\nplt.title('CatBoost Training and Validation Loss')\nplt.xlabel('Iterations')\nplt.ylabel('MAE')\nplt.legend()\nplt.grid()\nplt.show()\n\n\n# 預測與評估\ny_pred_cat = cat.predict(x_test)\nmae_cat = mean_absolute_error(y_test, y_pred_cat)\nprint(f\"CatBoost MAE: {mae_cat:.4f}\")","metadata":{"execution":{"iopub.status.busy":"2024-11-20T09:09:35.757290Z","iopub.execute_input":"2024-11-20T09:09:35.757683Z","iopub.status.idle":"2024-11-20T09:09:37.074857Z","shell.execute_reply.started":"2024-11-20T09:09:35.757649Z","shell.execute_reply":"2024-11-20T09:09:37.073715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 定義 sii_classify 函數\ndef sii_classify(data):\n    sii = []\n    for pred in data:\n        if pred >= 0 and pred <= 30:\n            sii.append(0)\n        elif pred > 30 and pred <= 50:\n            sii.append(1)\n        elif pred > 50 and pred <= 80:\n            sii.append(2)\n        elif pred > 80 and pred <= 100:\n            sii.append(3)\n    return sii","metadata":{"execution":{"iopub.status.busy":"2024-11-20T09:09:40.913995Z","iopub.execute_input":"2024-11-20T09:09:40.914401Z","iopub.status.idle":"2024-11-20T09:09:40.921021Z","shell.execute_reply.started":"2024-11-20T09:09:40.914362Z","shell.execute_reply":"2024-11-20T09:09:40.919677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 準備 test 資料，刪除不需要的列\nx_test_final = test.drop(columns=['id'])  # 與訓練集保持一致\n\n# 應用數據預處理管道\n#preprocessor.fit(x_test_final)  # 必須基於訓練集進行 fit\nx_test_final = preprocessor.transform(x_test_final)\n\n# LightGBM 預測\nlgbm_y_pred_test = lgbm.predict(x_test_final)\n\n# XGBoost 預測\nxgb_y_pred_test = xgb.predict(x_test_final)\n\n# CatBoost 預測\ncat_y_pred_test = cat.predict(x_test_final)\n\n# 使用 sii_classify 函數根據預測結果生成 sii 值\nlgbm_sii_values = sii_classify(lgbm_y_pred_test)\nxgb_sii_values = sii_classify(xgb_y_pred_test)\ncat_sii_values = sii_classify(cat_y_pred_test)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-20T09:10:16.043119Z","iopub.execute_input":"2024-11-20T09:10:16.043560Z","iopub.status.idle":"2024-11-20T09:10:16.141473Z","shell.execute_reply.started":"2024-11-20T09:10:16.043523Z","shell.execute_reply":"2024-11-20T09:10:16.140072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 將預測結果整合到 DataFrame\ntest_predictions = pd.DataFrame({\n    'id': test['id'],  # 保留 ID\n    'LightGBM': lgbm_y_pred_test,\n    'XGBoost': xgb_y_pred_test,\n    'CatBoost': cat_y_pred_test,\n    'LightGBM_sii': lgbm_sii_values,\n    'XGBoost_sii' : xgb_sii_values,\n    'CatBoost_sii' : cat_sii_values\n})\n\n\n# 查看結果\nprint(test_predictions.head(20))","metadata":{"execution":{"iopub.status.busy":"2024-11-20T09:13:45.866905Z","iopub.execute_input":"2024-11-20T09:13:45.867294Z","iopub.status.idle":"2024-11-20T09:13:45.878735Z","shell.execute_reply.started":"2024-11-20T09:13:45.867257Z","shell.execute_reply":"2024-11-20T09:13:45.877575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.stats import mode\n\nTotal_pred_sii = pd.DataFrame({\n    'lgbm': lgbm_sii_values,\n    'xgb': xgb_sii_values,\n    'cat': cat_sii_values\n})\n\nFinal_pred_sii = mode(Total_pred_sii, axis=1).mode.flatten()\n\nTotal_pred_sii['Final_sii'] = Final_pred_sii\nprint(Total_pred_sii)","metadata":{"execution":{"iopub.status.busy":"2024-11-20T09:18:34.687620Z","iopub.execute_input":"2024-11-20T09:18:34.688050Z","iopub.status.idle":"2024-11-20T09:18:34.704796Z","shell.execute_reply.started":"2024-11-20T09:18:34.688010Z","shell.execute_reply":"2024-11-20T09:18:34.703567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a DataFrame for the submission\n\nsubmission = pd.DataFrame({\n    'id' : test['id'],\n    'sii' : Total_pred_sii['Final_sii']\n})\n\nsubmission_file_path = 'submission.csv'\n\nsubmission.to_csv(submission_file_path, index = False)\n\nsubmission","metadata":{"execution":{"iopub.status.busy":"2024-11-20T09:21:54.349505Z","iopub.execute_input":"2024-11-20T09:21:54.350610Z","iopub.status.idle":"2024-11-20T09:21:54.370966Z","shell.execute_reply.started":"2024-11-20T09:21:54.350546Z","shell.execute_reply":"2024-11-20T09:21:54.369902Z"},"trusted":true},"execution_count":null,"outputs":[]}]}