{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":31236,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.impute import SimpleImputer\n\n# 1. 載入數據\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\n# 2. 排除數據洩漏 (Data Leakage)：果斷移除所有 PCIAT 相關特徵 [cite: 5]\n# 避免模型陷入「用答案預測答案」的邏輯謬誤 [cite: 5]\npciat_cols = [col for col in train.columns if 'PCIAT' in col]\ntrain = train.drop(columns=pciat_cols)\ntest = test.drop(columns=[col for col in pciat_cols if col in test.columns])\n\n# 3. 特徵合併邏輯：處理 PAQ 互補性 [cite: 7, 8]\n# 基於 PAQ_A (青少年) 與 PAQ_C (兒童) 的高度互補性進行整合 [cite: 7]\ndef merge_paq_features(df):\n    # Kaggle 原始欄位通常為 'PAQ_A-PAQ_A_Total' 與 'PAQ_C-PAQ_C_Total'\n    col_a = 'PAQ_A-PAQ_A_Total'\n    col_c = 'PAQ_C-PAQ_C_Total'\n    \n    if col_a in df.columns and col_c in df.columns:\n        # 參考課堂所學之特徵合併方法，整合為統一的運動量因子 [cite: 9]\n        df['PAQ_Total'] = df[col_a].combine_first(df[col_c])\n        return df.drop(columns=[col_a, col_c])\n    return df\n\ntrain = merge_paq_features(train)\ntest = merge_paq_features(test)\n\n# 4. 數據預處理：基於統計學原理的中位數填補 [cite: 10, 11]\n# 針對 BMI (偏度 1.63) 等高度偏態分佈，採用中位數填補以提升抗異常值能力 [cite: 13, 49, 50]\ntarget = 'sii'\ntrain_clean = train.dropna(subset=[target])\ny = train_clean[target]\nX = train_clean.drop(columns=['id', target])\nX_test = test.drop(columns=['id'])\n\n# 僅選擇數值型特徵進行運算\nX_numeric = X.select_dtypes(include=[np.number])\nX_test_numeric = X_test.select_dtypes(include=[np.number])\n\n# 實作 Exp_5 策略：中位數填補 [cite: 108, 111]\nimputer = SimpleImputer(strategy='median')\nX_imputed = imputer.fit_transform(X_numeric)\nX_test_imputed = imputer.transform(X_test_numeric)\n\n# 5. 模型建立：隨機森林回歸策略 [cite: 72]\n# 捕捉連續程度比強行切分標籤更能滿足 QWK 的平方級懲罰要求 [cite: 71, 73]\n# 根據 U 型誤差曲線，選擇 Max Depth = 5 以對抗過擬合 [cite: 119, 132, 136]\nmodel = RandomForestRegressor(\n    n_estimators=100,\n    max_depth=5,\n    random_state=42\n)\nmodel.fit(X_imputed, y)\n\n# 6. 經驗驅動的門檻優化 (Threshold Tuning) [cite: 87]\n# 經消融實驗證實，門檻設定在 0.7 時可達到最佳 QWK 分數 [cite: 89, 103]\nraw_preds = model.predict(X_test_imputed)\n\ndef apply_optimized_threshold(preds, threshold=0.7):\n    # 透過數據驅動的優化，平衡過度預測與保守預測 [cite: 90]\n    return np.clip(np.round(preds - (threshold - 0.5)).astype(int), 0, 3)\n\nfinal_preds = apply_optimized_threshold(raw_preds, threshold=0.7)\n\n# 7. 產生提交檔案\nsubmission = pd.DataFrame({\n    'id': test['id'],\n    'sii': final_preds\n})\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"第 21 組 Exp_5 預測流程執行完畢。\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-24T09:01:20.845333Z","iopub.execute_input":"2025-12-24T09:01:20.845985Z","iopub.status.idle":"2025-12-24T09:01:23.171818Z","shell.execute_reply.started":"2025-12-24T09:01:20.845957Z","shell.execute_reply":"2025-12-24T09:01:23.171138Z"}},"outputs":[],"execution_count":null}]}