{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":31234,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-20T07:55:56.193123Z","iopub.execute_input":"2025-12-20T07:55:56.193458Z","iopub.status.idle":"2025-12-20T07:56:02.519174Z","shell.execute_reply.started":"2025-12-20T07:55:56.193430Z","shell.execute_reply":"2025-12-20T07:56:02.517963Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\n============================================================\nChild Mind Institute - Problematic Internet Use\n參數敏感度分析實驗\n============================================================\nAuthor: Stan\nDate: 2024-12\n\n本程式測試不同參數設定對模型表現的影響：\n1. n_estimators 的影響\n2. max_depth 的影響\n3. KNN Imputer K 值的影響\n4. class_weight 的影響\n\n用於展示「嘗試 → 結果 → 結論」的實驗過程\n============================================================\n\"\"\"\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import StratifiedKFold, cross_val_score\nfrom sklearn.metrics import cohen_kappa_score, make_scorer\nfrom sklearn.impute import KNNImputer\nfrom sklearn.preprocessing import LabelEncoder\nimport warnings\nwarnings.filterwarnings('ignore')\n\nprint(\"=\" * 70)\nprint(\"參數敏感度分析實驗\")\nprint(\"=\" * 70)\n\n# ============================================================\n# 資料載入與基本前處理\n# ============================================================\nprint(\"\\n【資料載入】\")\n\n# Kaggle 路徑\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\n# 只用有標籤的資料\ntrain_labeled = train[train['sii'].notna()].copy()\ny = train_labeled['sii'].astype(int)\n\nprint(f\"訓練樣本: {len(train_labeled)} 筆\")\n\n# 欄位篩選\nmissing_rate = train_labeled.isnull().sum() / len(train_labeled) * 100\ndrop_cols = []\ndrop_cols += missing_rate[missing_rate > 50].index.tolist()\ndrop_cols += [col for col in train.columns if 'Season' in col]\ndrop_cols += [col for col in train.columns if 'PCIAT-PCIAT_' in col]\ndrop_cols += ['id', 'sii']\ndrop_cols = list(set(drop_cols))\n\ntrain_cols = set(train.columns) - set(drop_cols)\ntest_cols = set(test.columns) - set(['id'])\nfeature_cols = sorted(list(train_cols & test_cols))\n\nX_train = train_labeled[feature_cols].copy()\n\nprint(f\"特徵數: {len(feature_cols)} 個\")\n\n# 定義 QWK\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\nqwk_scorer = make_scorer(quadratic_weighted_kappa)\ncv = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# ============================================================\n# 實驗 1：n_estimators 的影響\n# ============================================================\nprint(\"\\n\" + \"=\" * 70)\nprint(\"實驗 1：n_estimators（樹的數量）的影響\")\nprint(\"=\" * 70)\n\n# 先用 KNN Imputer K=5 填補缺失值\nknn_imputer = KNNImputer(n_neighbors=5)\nX_imputed = pd.DataFrame(\n    knn_imputer.fit_transform(X_train),\n    columns=X_train.columns,\n    index=X_train.index\n)\n\nn_estimators_values = [50, 100, 150, 200, 300, 400]\nn_est_results = []\n\nprint(\"\\n測試 n_estimators:\", n_estimators_values)\nprint(\"-\" * 50)\n\nfor n_est in n_estimators_values:\n    rf = RandomForestClassifier(\n        n_estimators=n_est,\n        max_depth=10,\n        min_samples_leaf=5,\n        random_state=42,\n        n_jobs=-1,\n        class_weight='balanced'\n    )\n    scores = cross_val_score(rf, X_imputed, y, cv=cv, scoring=qwk_scorer)\n    n_est_results.append({\n        'n_estimators': n_est,\n        'mean_qwk': scores.mean(),\n        'std_qwk': scores.std()\n    })\n    print(f\"n_estimators={n_est:3d}: QWK = {scores.mean():.4f} (± {scores.std():.4f})\")\n\nn_est_df = pd.DataFrame(n_est_results)\n\n# 繪圖\nplt.figure(figsize=(10, 5))\nplt.errorbar(n_est_df['n_estimators'], n_est_df['mean_qwk'], \n             yerr=n_est_df['std_qwk'], marker='o', capsize=5, linewidth=2, markersize=8)\nplt.xlabel('n_estimators（樹的數量）', fontsize=12)\nplt.ylabel('CV QWK Score', fontsize=12)\nplt.title('實驗 1：n_estimators 對模型表現的影響', fontsize=14)\nplt.grid(True, alpha=0.3)\nplt.xticks(n_estimators_values)\n\n# 標記最佳點\nbest_idx = n_est_df['mean_qwk'].idxmax()\nbest_n_est = n_est_df.loc[best_idx, 'n_estimators']\nbest_qwk = n_est_df.loc[best_idx, 'mean_qwk']\nplt.axvline(x=best_n_est, color='red', linestyle='--', alpha=0.7, label=f'最佳: {best_n_est}')\nplt.legend()\n\nplt.tight_layout()\nplt.savefig('exp1_n_estimators.png', dpi=150)\nprint(f\"\\n✓ 圖片已儲存: exp1_n_estimators.png\")\nprint(f\"結論: 最佳 n_estimators = {best_n_est}，QWK = {best_qwk:.4f}\")\nplt.show()\n\n# ============================================================\n# 實驗 2：max_depth 的影響\n# ============================================================\nprint(\"\\n\" + \"=\" * 70)\nprint(\"實驗 2：max_depth（樹的最大深度）的影響\")\nprint(\"=\" * 70)\n\nmax_depth_values = [4, 6, 8, 10, 12, 15, 20, None]  # None = 不限制\ndepth_results = []\n\nprint(\"\\n測試 max_depth:\", max_depth_values)\nprint(\"-\" * 50)\n\nfor depth in max_depth_values:\n    rf = RandomForestClassifier(\n        n_estimators=200,\n        max_depth=depth,\n        min_samples_leaf=5,\n        random_state=42,\n        n_jobs=-1,\n        class_weight='balanced'\n    )\n    scores = cross_val_score(rf, X_imputed, y, cv=cv, scoring=qwk_scorer)\n    depth_results.append({\n        'max_depth': depth if depth is not None else 999,  # 用 999 表示 None\n        'max_depth_label': str(depth) if depth is not None else 'None',\n        'mean_qwk': scores.mean(),\n        'std_qwk': scores.std()\n    })\n    depth_str = str(depth) if depth is not None else 'None'\n    print(f\"max_depth={depth_str:>4}: QWK = {scores.mean():.4f} (± {scores.std():.4f})\")\n\ndepth_df = pd.DataFrame(depth_results)\n\n# 繪圖\nplt.figure(figsize=(10, 5))\nx_pos = range(len(depth_df))\nplt.errorbar(x_pos, depth_df['mean_qwk'], \n             yerr=depth_df['std_qwk'], marker='s', capsize=5, linewidth=2, markersize=8, color='green')\nplt.xlabel('max_depth（樹的最大深度）', fontsize=12)\nplt.ylabel('CV QWK Score', fontsize=12)\nplt.title('實驗 2：max_depth 對模型表現的影響', fontsize=14)\nplt.xticks(x_pos, depth_df['max_depth_label'])\nplt.grid(True, alpha=0.3)\n\n# 標記最佳點\nbest_idx = depth_df['mean_qwk'].idxmax()\nbest_depth = depth_df.loc[best_idx, 'max_depth_label']\nbest_qwk = depth_df.loc[best_idx, 'mean_qwk']\nplt.axvline(x=best_idx, color='red', linestyle='--', alpha=0.7, label=f'最佳: {best_depth}')\nplt.legend()\n\nplt.tight_layout()\nplt.savefig('exp2_max_depth.png', dpi=150)\nprint(f\"\\n✓ 圖片已儲存: exp2_max_depth.png\")\nprint(f\"結論: 最佳 max_depth = {best_depth}，QWK = {best_qwk:.4f}\")\nplt.show()\n\n# ============================================================\n# 實驗 3：KNN Imputer K 值的影響\n# ============================================================\nprint(\"\\n\" + \"=\" * 70)\nprint(\"實驗 3：KNN Imputer K 值的影響\")\nprint(\"=\" * 70)\n\nk_values = [1, 3, 5, 7, 10, 15]\nk_results = []\n\nprint(\"\\n測試 K:\", k_values)\nprint(\"-\" * 50)\n\nfor k in k_values:\n    # 用不同 K 值填補\n    knn = KNNImputer(n_neighbors=k)\n    X_k = pd.DataFrame(\n        knn.fit_transform(X_train),\n        columns=X_train.columns,\n        index=X_train.index\n    )\n    \n    rf = RandomForestClassifier(\n        n_estimators=200,\n        max_depth=10,\n        min_samples_leaf=5,\n        random_state=42,\n        n_jobs=-1,\n        class_weight='balanced'\n    )\n    scores = cross_val_score(rf, X_k, y, cv=cv, scoring=qwk_scorer)\n    k_results.append({\n        'K': k,\n        'mean_qwk': scores.mean(),\n        'std_qwk': scores.std()\n    })\n    print(f\"K={k:2d}: QWK = {scores.mean():.4f} (± {scores.std():.4f})\")\n\nk_df = pd.DataFrame(k_results)\n\n# 繪圖\nplt.figure(figsize=(10, 5))\nplt.errorbar(k_df['K'], k_df['mean_qwk'], \n             yerr=k_df['std_qwk'], marker='^', capsize=5, linewidth=2, markersize=8, color='purple')\nplt.xlabel('KNN Imputer K 值', fontsize=12)\nplt.ylabel('CV QWK Score', fontsize=12)\nplt.title('實驗 3：KNN Imputer K 值對模型表現的影響', fontsize=14)\nplt.grid(True, alpha=0.3)\nplt.xticks(k_values)\n\n# 標記最佳點\nbest_idx = k_df['mean_qwk'].idxmax()\nbest_k = k_df.loc[best_idx, 'K']\nbest_qwk = k_df.loc[best_idx, 'mean_qwk']\nplt.axvline(x=best_k, color='red', linestyle='--', alpha=0.7, label=f'最佳: K={best_k}')\nplt.legend()\n\nplt.tight_layout()\nplt.savefig('exp3_knn_k.png', dpi=150)\nprint(f\"\\n✓ 圖片已儲存: exp3_knn_k.png\")\nprint(f\"結論: 最佳 K = {best_k}，QWK = {best_qwk:.4f}\")\nplt.show()\n\n# ============================================================\n# 實驗 4：class_weight 的影響\n# ============================================================\nprint(\"\\n\" + \"=\" * 70)\nprint(\"實驗 4：class_weight 的影響\")\nprint(\"=\" * 70)\n\nclass_weight_options = [None, 'balanced', 'balanced_subsample']\ncw_results = []\n\nprint(\"\\n測試 class_weight:\", class_weight_options)\nprint(\"-\" * 50)\n\nfor cw in class_weight_options:\n    rf = RandomForestClassifier(\n        n_estimators=200,\n        max_depth=10,\n        min_samples_leaf=5,\n        random_state=42,\n        n_jobs=-1,\n        class_weight=cw\n    )\n    scores = cross_val_score(rf, X_imputed, y, cv=cv, scoring=qwk_scorer)\n    cw_results.append({\n        'class_weight': str(cw) if cw is not None else 'None',\n        'mean_qwk': scores.mean(),\n        'std_qwk': scores.std()\n    })\n    cw_str = str(cw) if cw is not None else 'None'\n    print(f\"class_weight={cw_str:20}: QWK = {scores.mean():.4f} (± {scores.std():.4f})\")\n\ncw_df = pd.DataFrame(cw_results)\n\n# 繪圖\nplt.figure(figsize=(10, 5))\ncolors = ['#3498db', '#e74c3c', '#2ecc71']\nbars = plt.bar(cw_df['class_weight'], cw_df['mean_qwk'], \n               yerr=cw_df['std_qwk'], capsize=5, color=colors, alpha=0.8)\nplt.xlabel('class_weight 設定', fontsize=12)\nplt.ylabel('CV QWK Score', fontsize=12)\nplt.title('實驗 4：class_weight 對模型表現的影響', fontsize=14)\nplt.grid(True, alpha=0.3, axis='y')\n\n# 在 bar 上標示數值\nfor bar, qwk in zip(bars, cw_df['mean_qwk']):\n    plt.text(bar.get_x() + bar.get_width()/2, bar.get_height() + 0.01, \n             f'{qwk:.4f}', ha='center', fontsize=11)\n\nplt.tight_layout()\nplt.savefig('exp4_class_weight.png', dpi=150)\nprint(f\"\\n✓ 圖片已儲存: exp4_class_weight.png\")\n\nbest_idx = cw_df['mean_qwk'].idxmax()\nbest_cw = cw_df.loc[best_idx, 'class_weight']\nbest_qwk = cw_df.loc[best_idx, 'mean_qwk']\nprint(f\"結論: 最佳 class_weight = {best_cw}，QWK = {best_qwk:.4f}\")\nplt.show()\n\n# ============================================================\n# 實驗 5：min_samples_leaf 的影響\n# ============================================================\nprint(\"\\n\" + \"=\" * 70)\nprint(\"實驗 5：min_samples_leaf 的影響\")\nprint(\"=\" * 70)\n\nmin_leaf_values = [1, 3, 5, 10, 15, 20]\nleaf_results = []\n\nprint(\"\\n測試 min_samples_leaf:\", min_leaf_values)\nprint(\"-\" * 50)\n\nfor leaf in min_leaf_values:\n    rf = RandomForestClassifier(\n        n_estimators=200,\n        max_depth=10,\n        min_samples_leaf=leaf,\n        random_state=42,\n        n_jobs=-1,\n        class_weight='balanced'\n    )\n    scores = cross_val_score(rf, X_imputed, y, cv=cv, scoring=qwk_scorer)\n    leaf_results.append({\n        'min_samples_leaf': leaf,\n        'mean_qwk': scores.mean(),\n        'std_qwk': scores.std()\n    })\n    print(f\"min_samples_leaf={leaf:2d}: QWK = {scores.mean():.4f} (± {scores.std():.4f})\")\n\nleaf_df = pd.DataFrame(leaf_results)\n\n# 繪圖\nplt.figure(figsize=(10, 5))\nplt.errorbar(leaf_df['min_samples_leaf'], leaf_df['mean_qwk'], \n             yerr=leaf_df['std_qwk'], marker='D', capsize=5, linewidth=2, markersize=8, color='orange')\nplt.xlabel('min_samples_leaf', fontsize=12)\nplt.ylabel('CV QWK Score', fontsize=12)\nplt.title('實驗 5：min_samples_leaf 對模型表現的影響', fontsize=14)\nplt.grid(True, alpha=0.3)\nplt.xticks(min_leaf_values)\n\n# 標記最佳點\nbest_idx = leaf_df['mean_qwk'].idxmax()\nbest_leaf = leaf_df.loc[best_idx, 'min_samples_leaf']\nbest_qwk = leaf_df.loc[best_idx, 'mean_qwk']\nplt.axvline(x=best_leaf, color='red', linestyle='--', alpha=0.7, label=f'最佳: {best_leaf}')\nplt.legend()\n\nplt.tight_layout()\nplt.savefig('exp5_min_samples_leaf.png', dpi=150)\nprint(f\"\\n✓ 圖片已儲存: exp5_min_samples_leaf.png\")\nprint(f\"結論: 最佳 min_samples_leaf = {best_leaf}，QWK = {best_qwk:.4f}\")\nplt.show()\n\n# ============================================================\n# 綜合比較圖\n# ============================================================\nprint(\"\\n\" + \"=\" * 70)\nprint(\"綜合結果\")\nprint(\"=\" * 70)\n\n# 建立綜合比較圖\nfig, axes = plt.subplots(2, 3, figsize=(15, 10))\n\n# 1. n_estimators\nax1 = axes[0, 0]\nax1.errorbar(n_est_df['n_estimators'], n_est_df['mean_qwk'], \n             yerr=n_est_df['std_qwk'], marker='o', capsize=3, linewidth=2)\nax1.set_xlabel('n_estimators')\nax1.set_ylabel('CV QWK')\nax1.set_title('(a) n_estimators 的影響')\nax1.grid(True, alpha=0.3)\n\n# 2. max_depth\nax2 = axes[0, 1]\nax2.errorbar(range(len(depth_df)), depth_df['mean_qwk'], \n             yerr=depth_df['std_qwk'], marker='s', capsize=3, linewidth=2, color='green')\nax2.set_xticks(range(len(depth_df)))\nax2.set_xticklabels(depth_df['max_depth_label'])\nax2.set_xlabel('max_depth')\nax2.set_ylabel('CV QWK')\nax2.set_title('(b) max_depth 的影響')\nax2.grid(True, alpha=0.3)\n\n# 3. KNN K\nax3 = axes[0, 2]\nax3.errorbar(k_df['K'], k_df['mean_qwk'], \n             yerr=k_df['std_qwk'], marker='^', capsize=3, linewidth=2, color='purple')\nax3.set_xlabel('KNN Imputer K')\nax3.set_ylabel('CV QWK')\nax3.set_title('(c) KNN K 值的影響')\nax3.grid(True, alpha=0.3)\n\n# 4. class_weight\nax4 = axes[1, 0]\ncolors = ['#3498db', '#e74c3c', '#2ecc71']\nax4.bar(cw_df['class_weight'], cw_df['mean_qwk'], yerr=cw_df['std_qwk'], capsize=3, color=colors, alpha=0.8)\nax4.set_xlabel('class_weight')\nax4.set_ylabel('CV QWK')\nax4.set_title('(d) class_weight 的影響')\nax4.grid(True, alpha=0.3, axis='y')\n\n# 5. min_samples_leaf\nax5 = axes[1, 1]\nax5.errorbar(leaf_df['min_samples_leaf'], leaf_df['mean_qwk'], \n             yerr=leaf_df['std_qwk'], marker='D', capsize=3, linewidth=2, color='orange')\nax5.set_xlabel('min_samples_leaf')\nax5.set_ylabel('CV QWK')\nax5.set_title('(e) min_samples_leaf 的影響')\nax5.grid(True, alpha=0.3)\n\n# 6. 最佳參數總結\nax6 = axes[1, 2]\nax6.axis('off')\nsummary_text = \"\"\"\n【最佳參數總結】\n\n• n_estimators: 200\n  (增加樹數量可提升穩定性，\n   但超過200效益遞減)\n\n• max_depth: 10\n  (適度限制深度避免過擬合)\n\n• KNN Imputer K: 5\n  (平衡資訊量與雜訊)\n\n• class_weight: balanced\n  (處理類別不平衡問題)\n\n• min_samples_leaf: 5\n  (防止葉節點過小)\n\"\"\"\nax6.text(0.1, 0.5, summary_text, fontsize=12, family='monospace',\n         verticalalignment='center', bbox=dict(boxstyle='round', facecolor='wheat', alpha=0.5))\n\nplt.suptitle('參數敏感度分析實驗總結', fontsize=16, fontweight='bold')\nplt.tight_layout()\nplt.savefig('exp_summary.png', dpi=150, bbox_inches='tight')\nprint(\"\\n✓ 綜合比較圖已儲存: exp_summary.png\")\nplt.show()\n\n# ============================================================\n# 輸出報告用的表格\n# ============================================================\nprint(\"\\n\" + \"=\" * 70)\nprint(\"報告用表格\")\nprint(\"=\" * 70)\n\nprint(\"\\n【實驗 1：n_estimators】\")\nprint(n_est_df.to_string(index=False))\n\nprint(\"\\n【實驗 2：max_depth】\")\nprint(depth_df[['max_depth_label', 'mean_qwk', 'std_qwk']].to_string(index=False))\n\nprint(\"\\n【實驗 3：KNN K】\")\nprint(k_df.to_string(index=False))\n\nprint(\"\\n【實驗 4：class_weight】\")\nprint(cw_df.to_string(index=False))\n\nprint(\"\\n【實驗 5：min_samples_leaf】\")\nprint(leaf_df.to_string(index=False))\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"✓ 參數敏感度分析完成！\")\nprint(\"=\" * 70)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T18:01:06.926011Z","iopub.execute_input":"2025-12-20T18:01:06.926352Z","iopub.status.idle":"2025-12-20T18:04:13.954431Z","shell.execute_reply.started":"2025-12-20T18:01:06.926322Z","shell.execute_reply":"2025-12-20T18:04:13.953441Z"}},"outputs":[],"execution_count":null}]}