{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1.引入包","metadata":{"editable":false}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nfrom scipy import stats\nimport numpy as np\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler\n\n# 忽略警告\nwarnings.simplefilter('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-11-27T03:03:17.664370Z","iopub.execute_input":"2024-11-27T03:03:17.664870Z","iopub.status.idle":"2024-11-27T03:03:17.670593Z","shell.execute_reply.started":"2024-11-27T03:03:17.664830Z","shell.execute_reply":"2024-11-27T03:03:17.669415Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2.录入数据","metadata":{"editable":false}},{"cell_type":"code","source":"locate = r'/kaggle/input/child-mind-institute-problematic-internet-use/train.csv'\ntest_loc = r'/kaggle/input/child-mind-institute-problematic-internet-use/test.csv'\n\ndata = pd.read_csv(locate)\n\ntest = pd.read_csv(test_loc)\n\ndata.set_index('id', inplace=True)\n\ntest.set_index('id',inplace=True)\n\ndata","metadata":{"execution":{"iopub.status.busy":"2024-11-27T03:03:17.680153Z","iopub.execute_input":"2024-11-27T03:03:17.680546Z","iopub.status.idle":"2024-11-27T03:03:17.767826Z","shell.execute_reply.started":"2024-11-27T03:03:17.680509Z","shell.execute_reply":"2024-11-27T03:03:17.766746Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2024-11-27T03:03:17.769727Z","iopub.execute_input":"2024-11-27T03:03:17.770197Z","iopub.status.idle":"2024-11-27T03:03:17.808478Z","shell.execute_reply.started":"2024-11-27T03:03:17.770148Z","shell.execute_reply":"2024-11-27T03:03:17.807231Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 获取 data 和 test 数据框的列名\ndata_columns = set(data.columns)\ntest_columns = set(test.columns)\n\n# 计算 data 中有而 test 中没有的列，排除 'sii'\ndata_not_in_test = data_columns - test_columns - {'sii'}\n\n# 计算 test 中有而 data 中没有的列\ntest_not_in_data = test_columns - data_columns\n\n\nprint(\"Columns in data but not in test:\", data_not_in_test)\nprint(\"Columns in test but not in data:\", test_not_in_data)\n\n# 删除 data 中存在但 test 中没有的列，且不删除 'sii'\ndata = data.drop(columns=data_not_in_test)\n\nprint(f\"删除的列: {data_not_in_test}\")\n\n# 输出删除的列数\nprint(f\"删除的列数: {len(data_not_in_test)}\")","metadata":{"execution":{"iopub.status.busy":"2024-11-27T03:03:17.809706Z","iopub.execute_input":"2024-11-27T03:03:17.810081Z","iopub.status.idle":"2024-11-27T03:03:17.820364Z","shell.execute_reply.started":"2024-11-27T03:03:17.810016Z","shell.execute_reply":"2024-11-27T03:03:17.819238Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2024-11-27T03:03:17.822849Z","iopub.execute_input":"2024-11-27T03:03:17.823337Z","iopub.status.idle":"2024-11-27T03:03:17.850311Z","shell.execute_reply.started":"2024-11-27T03:03:17.823285Z","shell.execute_reply":"2024-11-27T03:03:17.849223Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"总输入3960，出现不少缺省值，需要删去或者使用平均值代替。\n\n","metadata":{"editable":false}},{"cell_type":"markdown","source":"# 分析每一个column\n\n- 删除value少于30%的列\n- 删除每一行中缺省值大于30%\n- 将剩下的具有缺省值的数字列赋值\n\n**问题** object列的缺省值如何处理","metadata":{"editable":false}},{"cell_type":"markdown","source":"# 分析每一个column\n\n- 删除value少于70%的列\n- 删除每一行中缺省值大于30%\n- 将剩下的具有缺省值的数字列赋值\n\n**问题** object列的缺省值如何处理","metadata":{}},{"cell_type":"markdown","source":"## 删除缺省值过多的列","metadata":{"editable":false}},{"cell_type":"code","source":"# 删除 sii 列中 NaN 的行\ndata = data.dropna(subset=['sii'])\n\n# 计算每列缺省值的比例\nmissing_ratio = data.isnull().mean()\n\n# 找出缺失值比例超过70%的列 70可能删的真不少\n#columns_to_drop = missing_ratio[missing_ratio > 0.7].index\ncolumns_to_drop = missing_ratio[missing_ratio > 0.8].index\n\n# 删除缺失值比例超过70%的列\n#data = data.loc[:, missing_ratio <= 0.7]\n#test = test.loc[:, missing_ratio <= 0.7]\ndata = data.loc[:, missing_ratio <= 0.8]\ntest = test.loc[:, missing_ratio <= 0.8]\n\n# 输出被删除的列\nprint(\"被删除的列:\", list(columns_to_drop))","metadata":{"execution":{"iopub.status.busy":"2024-11-27T03:03:17.852312Z","iopub.execute_input":"2024-11-27T03:03:17.852743Z","iopub.status.idle":"2024-11-27T03:03:17.873268Z","shell.execute_reply.started":"2024-11-27T03:03:17.852693Z","shell.execute_reply":"2024-11-27T03:03:17.871967Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 删除某一行缺省值过度的行\n\n删除之后发现删除数量过多，不可行，导致欠拟合，所以直接进行下一步","metadata":{"editable":false}},{"cell_type":"markdown","source":"## 异常值处理","metadata":{"editable":false}},{"cell_type":"markdown","source":"### 1.箱线图分析数据分布","metadata":{"editable":false}},{"cell_type":"code","source":"# 1. 将 data 中为数值类型的放入 data_num\ndata_num = data.select_dtypes(include=['float64', 'int64'])\n\n# 2. 对 data_num 进行删除异常值处理\n\n## 设置 Z 分数阈值\nthreshold = 3  # 设置 Z 分数阈值\nz_scores = np.abs(stats.zscore(data_num))\n\n# 找出异常值所在的行\noutliers_zscore = (z_scores > threshold)\ndeleted_rows_zscore = np.where(outliers_zscore)[0]\nprint(f\"通过 Z 分数删除的行数: {len(deleted_rows_zscore)}\")\n\n# 删除包含 Z 分数异常值的行\ndata_num_clean_zscore = data_num[(~outliers_zscore).all(axis=1)]\n\n## 设定异常值范围：删除小于 0 或大于 10000 的值\nlower_bound = 0\nupper_bound = 12000\n\n# 找出超出范围的异常值\noutliers_range = (data_num_clean_zscore < lower_bound) | (data_num_clean_zscore > upper_bound)\n\n# 找出包含异常值的行\noutlier_rows_range = outliers_range.any(axis=1)\nprint(f\"通过超出范围删除的行数: {outlier_rows_range.sum()}\")\n\n# 删除超出范围的异常值行\ndata_num_clean_final = data_num_clean_zscore[~outlier_rows_range]\n\n# 3. 绘制删除异常值后的箱线图\nplt.figure(figsize=(15, 10))\nsns.boxplot(data=data_num_clean_final)\nplt.xticks(rotation=90)  # 旋转 x 轴标签便于阅读\nplt.title('删除异常值后的数据分布的箱线图')\nplt.show()\n\n# 输出被删除的行的索引\ndeleted_rows_range = np.where(outlier_rows_range)[0]\nprint(f\"被删除的行的索引 (Z 分数和超出范围方法): {np.concatenate([deleted_rows_zscore, deleted_rows_range])}\")\n\n# 4. 从原始 data 删除相应的异常值行\n# 删除 data 中的异常值行，保留非数值列\n\n# 合并 Z 分数和超出范围方法删除的索引\nall_deleted_rows = np.concatenate([deleted_rows_zscore, deleted_rows_range])\n\n# 找出实际存在于 data.index 中的行，避免 KeyError\nexisting_rows = [idx for idx in all_deleted_rows if idx in data.index]\n\n# 删除实际存在的异常值行\ndata_clean = data.drop(index=existing_rows)\n\n# 输出确认\nprint(f\"成功删除的行数: {len(existing_rows)}\")\nprint(f\"清洗后的数据: \\n{data_clean.head()}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-27T03:03:17.875586Z","iopub.execute_input":"2024-11-27T03:03:17.876072Z","iopub.status.idle":"2024-11-27T03:03:18.967287Z","shell.execute_reply.started":"2024-11-27T03:03:17.876002Z","shell.execute_reply":"2024-11-27T03:03:18.966050Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_clean","metadata":{"execution":{"iopub.status.busy":"2024-11-27T03:03:18.968421Z","iopub.execute_input":"2024-11-27T03:03:18.968731Z","iopub.status.idle":"2024-11-27T03:03:19.003776Z","shell.execute_reply.started":"2024-11-27T03:03:18.968698Z","shell.execute_reply":"2024-11-27T03:03:19.002616Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test\n","metadata":{"execution":{"iopub.status.busy":"2024-11-27T03:03:19.006510Z","iopub.execute_input":"2024-11-27T03:03:19.006866Z","iopub.status.idle":"2024-11-27T03:03:19.044930Z","shell.execute_reply.started":"2024-11-27T03:03:19.006831Z","shell.execute_reply":"2024-11-27T03:03:19.043826Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 处理缺省值\n\n**方法**\n- 对于数值型，将所有缺省值按照当前列的中位数作为填充\n- 对于object，先观察缺省数量，如果过多，则使用unknown作为新类","metadata":{"editable":false}},{"cell_type":"code","source":"# 删除缺失值比例超过70%的行\ndata_clean = data_clean[data_clean.isnull().mean(axis=1) <= 0.8]\n\ndata_clean","metadata":{"execution":{"iopub.status.busy":"2024-11-27T03:03:19.046207Z","iopub.execute_input":"2024-11-27T03:03:19.046554Z","iopub.status.idle":"2024-11-27T03:03:19.086397Z","shell.execute_reply.started":"2024-11-27T03:03:19.046519Z","shell.execute_reply":"2024-11-27T03:03:19.084899Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\n# 获取所有数值型列，并排除分类标签 'sii'\nnum_cols = data_clean.select_dtypes(include=['float64', 'int64']).columns\nnum_cols = num_cols.drop('sii', errors='ignore')  # 移除 'sii' 列\n\n# 对数值型列填充缺省值，使用该列的平均值填充\ndata_clean[num_cols] = data_clean[num_cols].apply(lambda col: col.fillna(col.mean()))\ntest[num_cols] = test[num_cols].apply(lambda col: col.fillna(col.mean()))\n\n# 初始化标准化工具\nscaler = StandardScaler()\n\n# 对数值型列进行标准化\ndata_clean[num_cols] = scaler.fit_transform(data_clean[num_cols])\ntest[num_cols] = scaler.transform(test[num_cols])  # 仅变换测试集\n\n# 对类别型列填充缺省值\nfor col in data_clean.select_dtypes(include=['object']).columns:\n    missing_percentage = data_clean[col].isnull().mean()  # 计算缺省值的比例\n    if missing_percentage > 0.5:\n        # 如果缺省值超过50%，填充为 'unknown'\n        data_clean[col].fillna('unknown', inplace=True)\n    else:\n        # 否则使用该列的众数填充\n        data_clean[col].fillna(data_clean[col].mode()[0], inplace=True)\n\n# 将非标准缺失值（如空字符串或自定义标记）统一转换为 NaN\ntest.replace(['', 'missing', 'null', None], np.nan, inplace=True)\n\n# 然后继续处理缺省值\nfor col in test.select_dtypes(include=['object']).columns:\n    missing_percentage = test[col].isnull().mean()\n    if missing_percentage > 0.5:\n        test[col].fillna('unknown', inplace=True)\n    else:\n        test[col].fillna(test[col].mode()[0], inplace=True)\n\n\n\n# 输出结果以验证数据清洗和标准化过程是否成功\nprint(data_clean.head())\n","metadata":{"execution":{"iopub.status.busy":"2024-11-27T03:03:19.087630Z","iopub.execute_input":"2024-11-27T03:03:19.088032Z","iopub.status.idle":"2024-11-27T03:03:19.192862Z","shell.execute_reply.started":"2024-11-27T03:03:19.087997Z","shell.execute_reply":"2024-11-27T03:03:19.191748Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 检查是否还有缺省值\nprint(data_clean.isnull().sum())\n\ndata_clean","metadata":{"execution":{"iopub.status.busy":"2024-11-27T03:03:19.194284Z","iopub.execute_input":"2024-11-27T03:03:19.194711Z","iopub.status.idle":"2024-11-27T03:03:19.234919Z","shell.execute_reply.started":"2024-11-27T03:03:19.194662Z","shell.execute_reply":"2024-11-27T03:03:19.233670Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2024-11-27T03:03:19.236451Z","iopub.execute_input":"2024-11-27T03:03:19.236790Z","iopub.status.idle":"2024-11-27T03:03:19.266921Z","shell.execute_reply.started":"2024-11-27T03:03:19.236753Z","shell.execute_reply":"2024-11-27T03:03:19.265790Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 使用训练集的列名进行 One-hot 编码\ndata_clean_encoded = pd.get_dummies(data_clean, drop_first=True)\n\n# 确保训练集和测试集使用相同的编码规则\ntest_encoded = pd.get_dummies(test, drop_first=True)\n\n# 对测试集补齐训练集中有但测试集中缺失的列\nmissing_cols = set(data_clean_encoded.columns) - set(test_encoded.columns)\nfor col in missing_cols:\n    test_encoded[col] = 0  # 补充缺失列并赋值为0\n\n# 保证列的顺序与训练集一致\ntest_encoded = test_encoded[data_clean_encoded.columns]\n\n# 确保列类型保持一致（必要时强制转换）\nfor col in data_clean_encoded.select_dtypes(include=['object']).columns:\n    if col in test_encoded.columns:\n        test_encoded[col] = test_encoded[col].astype('object')\n","metadata":{"execution":{"iopub.status.busy":"2024-11-27T03:03:19.269475Z","iopub.execute_input":"2024-11-27T03:03:19.269820Z","iopub.status.idle":"2024-11-27T03:03:19.305105Z","shell.execute_reply.started":"2024-11-27T03:03:19.269785Z","shell.execute_reply":"2024-11-27T03:03:19.303852Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in data_clean.columns:\n    if col in test.columns:\n        if data_clean[col].dtype != test[col].dtype:\n            print(f\"列 {col} 类型不一致：训练集为 {data_clean[col].dtype}，测试集为 {test[col].dtype}\")","metadata":{"execution":{"iopub.status.busy":"2024-11-27T03:03:19.306364Z","iopub.execute_input":"2024-11-27T03:03:19.306683Z","iopub.status.idle":"2024-11-27T03:03:19.314835Z","shell.execute_reply.started":"2024-11-27T03:03:19.306649Z","shell.execute_reply":"2024-11-27T03:03:19.313658Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 特征工程\n\n- 使用特征集合和拆分的形式处理各个特征","metadata":{"editable":false}},{"cell_type":"code","source":"","metadata":{"editable":false,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 拟合模型\n\n**多分类**问题：需要找到可以适用于多分类模型来进行预测\n\n- 随机森林：data_for\n- 梯度提升决策树：data_GBT\n- Voting Classifier：data_vote\n- XGBoost / CatBoost: data_xg","metadata":{"editable":false}},{"cell_type":"markdown","source":"# 随机森林","metadata":{"editable":false}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split, cross_val_score\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.preprocessing import LabelEncoder, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.metrics import accuracy_score, confusion_matrix, precision_score, recall_score, f1_score\n\n# 1. 识别非数值列\ncategorical_cols = data_clean.select_dtypes(include=['object']).columns\n\n# 2. 使用 One-Hot 编码将非数值列转换为数值类型\n# 保留 sii 作为标签，不转换\nX = data_clean.drop(columns='sii')\ny = data_clean['sii']\n\n# 使用 LabelEncoder 对标签进行编码（如果标签是类别型的）\nlabel_encoder = LabelEncoder()\ny = label_encoder.fit_transform(y)\n\n# 构建一个 pipeline 对 categorical features 进行 One-Hot Encoding\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('cat', OneHotEncoder(handle_unknown='ignore', drop='first'), categorical_cols)],  # 使用 one-hot 编码转换非数值列\n    remainder='passthrough'  # 数值列保持不变\n)\n\n# 3. 构建包含预处理和模型的 pipeline\nrf_model = Pipeline(steps=[\n    ('preprocessor', preprocessor),\n    ('classifier', RandomForestClassifier(random_state=42))\n])\n\n# 4. 划分训练集和测试集\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42, stratify=y)\n\n# 5. 使用交叉验证来拟合模型\ncross_val_scores = cross_val_score(rf_model, X_train, y_train, cv=7, scoring='accuracy')\nprint(f\"交叉验证的准确率: {cross_val_scores.mean():.4f}\")\n\n# 6. 在训练集上训练模型\nrf_model.fit(X_train, y_train)\n\n# 7. 在测试集上进行预测\ny_pred = rf_model.predict(X_test)\n\n# 8. 评估模型\naccuracy = accuracy_score(y_test, y_pred)\nconf_matrix = confusion_matrix(y_test, y_pred)\nprecision = precision_score(y_test, y_pred, average='weighted')\nrecall = recall_score(y_test, y_pred, average='weighted')\nf1 = f1_score(y_test, y_pred, average='weighted')\n\nprint(f\"准确率: {accuracy:.4f}\")\nprint(\"混淆矩阵:\\n\", conf_matrix)\nprint(f\"精确率: {precision:.4f}\")\nprint(f\"召回率: {recall:.4f}\")\nprint(f\"F1分数: {f1:.4f}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-27T03:03:19.316238Z","iopub.execute_input":"2024-11-27T03:03:19.316584Z","iopub.status.idle":"2024-11-27T03:03:24.928850Z","shell.execute_reply.started":"2024-11-27T03:03:19.316539Z","shell.execute_reply":"2024-11-27T03:03:24.927724Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 随机森林拟合效果\n\n通过对预测结果的评估看出模型拟合的非常贴合，但是正因如此可能会产生大量过拟合数据，于是尝试多种交叉验证处理或者尝试别的模型","metadata":{"editable":false}},{"cell_type":"markdown","source":"# 梯度提升决策树","metadata":{"editable":false}},{"cell_type":"code","source":"data_GBT = data_clean\n\nfrom sklearn.model_selection import train_test_split, cross_val_score\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import accuracy_score, confusion_matrix, precision_score, recall_score, f1_score\nimport pandas as pd\n\n# 1. 处理类别型数据\ndata_encoded = pd.get_dummies(data_GBT, drop_first=True)  # One-hot 编码类别型变量\n\n# 2. 划分特征和目标列\nX = data_encoded.drop(columns=['sii'])  # 假设 'sii' 是目标列\ny = data_encoded['sii']\n\n# 3. 划分训练集和测试集\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# 4. 标准化特征\nscaler = StandardScaler()\nX_train = scaler.fit_transform(X_train)\nX_test = scaler.transform(X_test)\n\n# 5. 定义梯度提升决策树模型\ngb_model = GradientBoostingClassifier(random_state=42)\n\n# 6. 使用交叉验证评估模型\ncv_scores = cross_val_score(gb_model, X_train, y_train, cv=5, scoring='accuracy')\nprint(f\"交叉验证准确率: {cv_scores.mean():.4f}\")\n\n# 7. 训练模型\ngb_model.fit(X_train, y_train)\n\n# 8. 进行预测\ny_pred = gb_model.predict(X_test)\n\n# 9. 评估模型\naccuracy = accuracy_score(y_test, y_pred)\nconf_matrix = confusion_matrix(y_test, y_pred)\nprecision = precision_score(y_test, y_pred, average='weighted')\nrecall = recall_score(y_test, y_pred, average='weighted')\nf1 = f1_score(y_test, y_pred, average='weighted')\n\nprint(f\"准确率: {accuracy:.4f}\")\nprint(\"混淆矩阵:\")\nprint(conf_matrix)\nprint(f\"精确率: {precision:.4f}\")\nprint(f\"召回率: {recall:.4f}\")\nprint(f\"F1 分数: {f1:.4f}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-27T03:03:24.929989Z","iopub.execute_input":"2024-11-27T03:03:24.930321Z","iopub.status.idle":"2024-11-27T03:03:59.832306Z","shell.execute_reply.started":"2024-11-27T03:03:24.930288Z","shell.execute_reply":"2024-11-27T03:03:59.831101Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 梯度提升决策树拟合效果\n\n结果绝对过拟合，但是不知道怎么处理","metadata":{"editable":false}},{"cell_type":"markdown","source":"# Voting Classifier","metadata":{"editable":false}},{"cell_type":"code","source":"data_vote = data_clean\n\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier, VotingClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import train_test_split, cross_val_score\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import accuracy_score, confusion_matrix, precision_score, recall_score, f1_score\nimport pandas as pd\n\n# 1. 对类别型数据进行编码\ndata_encoded = pd.get_dummies(data_vote, drop_first=True)  # 对类别型变量进行 One-hot 编码\n\n# 2. 划分特征和目标列\nX = data_encoded.drop(columns=['sii'])  # 假设 'sii' 是目标列\ny = data_encoded['sii']\n\n# 3. 划分训练集和测试集\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# 4. 标准化特征\nscaler = StandardScaler()\nX_train = scaler.fit_transform(X_train)\nX_test = scaler.transform(X_test)\n\n# 5. 定义 Voting Classifier 模型\nvoting_clf = VotingClassifier(\n    estimators=[\n        ('rf', RandomForestClassifier(n_estimators=100, random_state=42)),\n        ('gb', GradientBoostingClassifier(n_estimators=100, random_state=42)),\n        ('lr', LogisticRegression(max_iter=200, random_state=42))\n    ],\n    voting='soft'  # 使用软投票\n    #voting='hard'  # 使用硬投票\n)\n\n# 6. 使用交叉验证评估模型\ncv_scores = cross_val_score(voting_clf, X_train, y_train, cv=5, scoring='accuracy')\nprint(f\"交叉验证准确率: {cv_scores.mean():.4f}\")\n\n# 7. 训练 Voting Classifier 模型\nvoting_clf.fit(X_train, y_train)\n\n# 8. 进行预测\ny_pred = voting_clf.predict(X_test)\n\n# 9. 评估模型\naccuracy = accuracy_score(y_test, y_pred)\nconf_matrix = confusion_matrix(y_test, y_pred)\nprecision = precision_score(y_test, y_pred, average='weighted')\nrecall = recall_score(y_test, y_pred, average='weighted')\nf1 = f1_score(y_test, y_pred, average='weighted')\n\nprint(f\"准确率: {accuracy:.4f}\")\nprint(\"混淆矩阵:\")\nprint(conf_matrix)\nprint(f\"精确率: {precision:.4f}\")\nprint(f\"召回率: {recall:.4f}\")\nprint(f\"F1 分数: {f1:.4f}\")","metadata":{"execution":{"iopub.status.busy":"2024-11-27T03:03:59.833650Z","iopub.execute_input":"2024-11-27T03:03:59.834004Z","iopub.status.idle":"2024-11-27T03:04:40.293081Z","shell.execute_reply.started":"2024-11-27T03:03:59.833968Z","shell.execute_reply":"2024-11-27T03:04:40.291184Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 10. 使用训练好的预处理器对 test 数据集进行转换\ndata_encoded_test = pd.get_dummies(test, drop_first=True) \n# 对 test 数据进行 One-hot 编码 \nX_test_transformed = data_encoded_test.reindex(columns=X.columns, fill_value=0) \n# 保证列与训练集一致 \nX_test_transformed = scaler.transform(X_test_transformed) # 使用训练集的标准化参数 \n# 11. 对转换后的数据进行预测\ny_pred_test = voting_clf.predict(X_test_transformed) \n# 12. 构建提交文件\nsubmission = pd.DataFrame({'id': test.index, 'sii': y_pred_test})\n# 13. 保存提交文件 \nsubmission.to_csv('submission.csv', index=False) \nprint(\"提交文件已保存为 'submission.csv'\")","metadata":{"execution":{"iopub.status.busy":"2024-11-27T03:04:40.294346Z","iopub.execute_input":"2024-11-27T03:04:40.294759Z","iopub.status.idle":"2024-11-27T03:04:40.339684Z","shell.execute_reply.started":"2024-11-27T03:04:40.294707Z","shell.execute_reply":"2024-11-27T03:04:40.337128Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#预测结果\ny_pred_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T03:04:40.342174Z","iopub.execute_input":"2024-11-27T03:04:40.345402Z","iopub.status.idle":"2024-11-27T03:04:40.356513Z","shell.execute_reply.started":"2024-11-27T03:04:40.345338Z","shell.execute_reply":"2024-11-27T03:04:40.354779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Voting Classifier拟合效果\n\n出现了 \n\n**过拟合！！！**","metadata":{"editable":false}},{"cell_type":"markdown","source":"# XGBoost","metadata":{"editable":false}},{"cell_type":"code","source":"data_XG = data_clean\n\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split, cross_val_score\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import accuracy_score, confusion_matrix, precision_score, recall_score, f1_score\nfrom xgboost import XGBClassifier\n\n# 1. 数据预处理\n# 将类别型数据进行 One-hot 编码\ndata_encoded = pd.get_dummies(data_XG, drop_first=True)\n\n# 划分特征和目标列\nX = data_encoded.drop(columns=['sii'])  # 假设 'sii' 是目标列\ny = data_encoded['sii']\n\n# 划分训练集和测试集\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# 标准化特征（可选）\nscaler = StandardScaler()\nX_train = scaler.fit_transform(X_train)\nX_test = scaler.transform(X_test)\n\n# 2. 定义 XGBoost 模型\nxgb_model = XGBClassifier(random_state=42)\n\n# 3. 交叉验证评估模型\ncv_scores = cross_val_score(xgb_model, X_train, y_train, cv=5, scoring='accuracy')\nprint(f\"交叉验证准确率: {cv_scores.mean():.4f}\")\n\n# 4. 训练模型\nxgb_model.fit(X_train, y_train)\n\n# 5. 预测\ny_pred = xgb_model.predict(X_test)\n\n# 6. 模型评估\naccuracy = accuracy_score(y_test, y_pred)\nconf_matrix = confusion_matrix(y_test, y_pred)\nprecision = precision_score(y_test, y_pred, average='weighted')\nrecall = recall_score(y_test, y_pred, average='weighted')\nf1 = f1_score(y_test, y_pred, average='weighted')\n\nprint(f\"准确率: {accuracy:.4f}\")\nprint(\"混淆矩阵:\")\nprint(conf_matrix)\nprint(f\"精确率: {precision:.4f}\")\nprint(f\"召回率: {recall:.4f}\")\nprint(f\"F1 分数: {f1:.4f}\")","metadata":{"execution":{"iopub.status.busy":"2024-11-27T03:04:40.358835Z","iopub.execute_input":"2024-11-27T03:04:40.359611Z","iopub.status.idle":"2024-11-27T03:04:45.269041Z","shell.execute_reply.started":"2024-11-27T03:04:40.359558Z","shell.execute_reply":"2024-11-27T03:04:45.267895Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## XGBoost预测效果\n\n完蛋了全部过拟合耶耶耶","metadata":{"editable":false}},{"cell_type":"markdown","source":"# 存储输出内容并且提交","metadata":{"editable":false}},{"cell_type":"code","source":"test_locate = r'/kaggle/input/child-mind-institute-problematic-internet-use/test.csv'","metadata":{"execution":{"iopub.status.busy":"2024-11-27T03:04:45.270384Z","iopub.execute_input":"2024-11-27T03:04:45.270703Z","iopub.status.idle":"2024-11-27T03:04:45.274832Z","shell.execute_reply.started":"2024-11-27T03:04:45.270668Z","shell.execute_reply":"2024-11-27T03:04:45.274018Z"},"trusted":true},"outputs":[],"execution_count":null}]}