{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":10245336,"sourceType":"datasetVersion","datasetId":6336300}],"dockerImageVersionId":30822,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import","metadata":{}},{"cell_type":"code","source":"import polars as pl\nimport pandas as pd\nimport numpy as np\n\nimport warnings\nwarnings.filterwarnings(\"ignore\") #忽略警告信息\n\nfrom sklearn.linear_model import LinearRegression\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n#from scipy.stats import chi2_contingency\n#from sklearn.preprocessing import PolynomialFeatures\n#from sklearn.feature_selection import SelectKBest, f_regression\n#from sklearn.model_selection import train_test_split\n#from sklearn.impute import SimpleImputer\nfrom sklearn.metrics import r2_score","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:45:41.714026Z","iopub.execute_input":"2024-12-29T09:45:41.714453Z","iopub.status.idle":"2024-12-29T09:45:44.522719Z","shell.execute_reply.started":"2024-12-29T09:45:41.714419Z","shell.execute_reply":"2024-12-29T09:45:44.521729Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"train = pl.scan_parquet(\n    f\"/kaggle/input/20241219-data/training.parquet\"\n).collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:45:44.523954Z","iopub.execute_input":"2024-12-29T09:45:44.524413Z","iopub.status.idle":"2024-12-29T09:45:53.811401Z","shell.execute_reply.started":"2024-12-29T09:45:44.524384Z","shell.execute_reply":"2024-12-29T09:45:53.810255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.to_pandas()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:45:53.813293Z","iopub.execute_input":"2024-12-29T09:45:53.813663Z","iopub.status.idle":"2024-12-29T09:45:58.763959Z","shell.execute_reply.started":"2024-12-29T09:45:53.813629Z","shell.execute_reply":"2024-12-29T09:45:58.762992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:45:58.765566Z","iopub.execute_input":"2024-12-29T09:45:58.765952Z","iopub.status.idle":"2024-12-29T09:45:58.804013Z","shell.execute_reply.started":"2024-12-29T09:45:58.765900Z","shell.execute_reply":"2024-12-29T09:45:58.803005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 查看feature_15","metadata":{}},{"cell_type":"code","source":"feature_15 = train[\"feature_15\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:45:58.804991Z","iopub.execute_input":"2024-12-29T09:45:58.805275Z","iopub.status.idle":"2024-12-29T09:45:58.809583Z","shell.execute_reply.started":"2024-12-29T09:45:58.805250Z","shell.execute_reply":"2024-12-29T09:45:58.808652Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_15.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:45:58.810819Z","iopub.execute_input":"2024-12-29T09:45:58.811217Z","iopub.status.idle":"2024-12-29T09:45:59.178420Z","shell.execute_reply.started":"2024-12-29T09:45:58.811178Z","shell.execute_reply":"2024-12-29T09:45:59.177077Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_15","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:45:59.179394Z","iopub.execute_input":"2024-12-29T09:45:59.179707Z","iopub.status.idle":"2024-12-29T09:45:59.187312Z","shell.execute_reply.started":"2024-12-29T09:45:59.179674Z","shell.execute_reply":"2024-12-29T09:45:59.186048Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"发现feature_15有NaN","metadata":{}},{"cell_type":"code","source":"responder_6 = train[\"responder_6\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:45:59.190271Z","iopub.execute_input":"2024-12-29T09:45:59.190614Z","iopub.status.idle":"2024-12-29T09:45:59.205566Z","shell.execute_reply.started":"2024-12-29T09:45:59.190582Z","shell.execute_reply":"2024-12-29T09:45:59.204238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"responder_6","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:45:59.207862Z","iopub.execute_input":"2024-12-29T09:45:59.208204Z","iopub.status.idle":"2024-12-29T09:45:59.227498Z","shell.execute_reply.started":"2024-12-29T09:45:59.208171Z","shell.execute_reply":"2024-12-29T09:45:59.226460Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"correlation = train.corr()['responder_6']\nprint(correlation['feature_15'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:45:59.228534Z","iopub.execute_input":"2024-12-29T09:45:59.228897Z","iopub.status.idle":"2024-12-29T09:49:17.777350Z","shell.execute_reply.started":"2024-12-29T09:45:59.228855Z","shell.execute_reply":"2024-12-29T09:49:17.776196Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"feature_15 与 responder_6 之间关联程度的强弱","metadata":{}},{"cell_type":"code","source":"#train['feature_15_binned'] = pd.qcut(train['feature_15'], q=4)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:49:17.778412Z","iopub.execute_input":"2024-12-29T09:49:17.778718Z","iopub.status.idle":"2024-12-29T09:49:17.782984Z","shell.execute_reply.started":"2024-12-29T09:49:17.778683Z","shell.execute_reply":"2024-12-29T09:49:17.781726Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 分析feature_15与responder_6的关系","metadata":{}},{"cell_type":"markdown","source":" 从图中可以看出，大部分数据点集中在feature_15接近 0 的区域。随着feature_15值的增加，数据点变得非常稀疏。responder_6的值分布在 - 4 到 4 之间，并且在feature_15为 0 附近有大量的数据点堆积","metadata":{}},{"cell_type":"code","source":"# 直方图\nsns.histplot(feature_15, kde=True)\nplt.title(\"feature_15 Distribution (histogram)\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:49:17.784075Z","iopub.execute_input":"2024-12-29T09:49:17.784460Z","iopub.status.idle":"2024-12-29T09:50:07.692024Z","shell.execute_reply.started":"2024-12-29T09:49:17.784421Z","shell.execute_reply":"2024-12-29T09:50:07.690793Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 箱线图\nsns.boxplot(feature_15)\nplt.title(\"feature_15 Boxplot\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:50:07.693177Z","iopub.execute_input":"2024-12-29T09:50:07.693589Z","iopub.status.idle":"2024-12-29T09:50:08.321069Z","shell.execute_reply.started":"2024-12-29T09:50:07.693544Z","shell.execute_reply":"2024-12-29T09:50:08.319843Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#散点图：原始特征与响应变量的关系\nsns.scatterplot(x='feature_15', y='responder_6', data=train)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:50:08.322245Z","iopub.execute_input":"2024-12-29T09:50:08.322649Z","iopub.status.idle":"2024-12-29T09:50:21.388522Z","shell.execute_reply.started":"2024-12-29T09:50:08.322609Z","shell.execute_reply":"2024-12-29T09:50:21.387280Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":" 查看缺失值","metadata":{}},{"cell_type":"code","source":"print(pd.Series(train['feature_15']).isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:50:21.389829Z","iopub.execute_input":"2024-12-29T09:50:21.390250Z","iopub.status.idle":"2024-12-29T09:50:21.404303Z","shell.execute_reply.started":"2024-12-29T09:50:21.390212Z","shell.execute_reply":"2024-12-29T09:50:21.403105Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"使用SimpleImputer填充缺失值","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:50:21.405449Z","iopub.execute_input":"2024-12-29T09:50:21.405868Z","iopub.status.idle":"2024-12-29T09:50:21.466874Z","shell.execute_reply.started":"2024-12-29T09:50:21.405836Z","shell.execute_reply":"2024-12-29T09:50:21.465635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"imputer = SimpleImputer(strategy='mean')\ntrain[['feature_15', 'responder_6']] = imputer.fit_transform(train[['feature_15', 'responder_6']])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:50:21.468179Z","iopub.execute_input":"2024-12-29T09:50:21.468504Z","iopub.status.idle":"2024-12-29T09:50:21.674799Z","shell.execute_reply.started":"2024-12-29T09:50:21.468477Z","shell.execute_reply":"2024-12-29T09:50:21.673511Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 创建平方项作为高阶特征\ntrain['feature_15_squared'] = train['feature_15'] ** 2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:50:21.676076Z","iopub.execute_input":"2024-12-29T09:50:21.676375Z","iopub.status.idle":"2024-12-29T09:50:21.740361Z","shell.execute_reply.started":"2024-12-29T09:50:21.676350Z","shell.execute_reply":"2024-12-29T09:50:21.739212Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 离散化feature_15\ntrain['feature_15_binned'] = pd.qcut(train['feature_15'], q=4, duplicates='drop').astype(str)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:50:21.741489Z","iopub.execute_input":"2024-12-29T09:50:21.741886Z","iopub.status.idle":"2024-12-29T09:50:24.368610Z","shell.execute_reply.started":"2024-12-29T09:50:21.741857Z","shell.execute_reply":"2024-12-29T09:50:24.367536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 相关性分析\nnumeric_features = train.select_dtypes(include=[np.number]).columns\ncorrelation = train[numeric_features].corr()['responder_6']\nprint(\"Correlation with responder_6:\")\nprint(correlation[['feature_15', 'feature_15_squared']])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:50:24.369760Z","iopub.execute_input":"2024-12-29T09:50:24.370220Z","iopub.status.idle":"2024-12-29T09:53:48.914956Z","shell.execute_reply.started":"2024-12-29T09:50:24.370178Z","shell.execute_reply":"2024-12-29T09:53:48.913770Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 可视化分析\nplt.figure(figsize=(12, 8))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:53:48.916055Z","iopub.execute_input":"2024-12-29T09:53:48.916360Z","iopub.status.idle":"2024-12-29T09:53:48.978801Z","shell.execute_reply.started":"2024-12-29T09:53:48.916325Z","shell.execute_reply":"2024-12-29T09:53:48.977827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 箱形图：离散化后的特征与响应变量的关系，数据被分成了四个区间，每个区间内的数据分布较为一致，异常值在各个区间内分布较为均匀\nplt.figure(figsize=(12, 10))\nplt.subplot(2, 2, 2)\nsns.boxplot(x='feature_15_binned', y='responder_6', data=train)\nplt.title('Box Plot of Binned feature_15 vs responder_6')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:53:48.979709Z","iopub.execute_input":"2024-12-29T09:53:48.980062Z","iopub.status.idle":"2024-12-29T09:53:54.394580Z","shell.execute_reply.started":"2024-12-29T09:53:48.980036Z","shell.execute_reply":"2024-12-29T09:53:54.393220Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 散点图：高阶特征与响应变量的关系\n#数据被平方后，数据点在各个区间内分布较为一致，异常值在较高的数值区域。\n#plt.subplot(2, 2, 3)\nsns.scatterplot(x='feature_15_squared', y='responder_6', data=train)\nplt.title('Scatter Plot of feature_15_squared vs responder_6')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:53:54.399147Z","iopub.execute_input":"2024-12-29T09:53:54.399475Z","iopub.status.idle":"2024-12-29T09:54:19.146893Z","shell.execute_reply.started":"2024-12-29T09:53:54.399447Z","shell.execute_reply":"2024-12-29T09:54:19.145608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 构建线性回归模型并评估R^2分数\nX = train[['feature_15', 'feature_15_squared']]\ny = train['responder_6']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:54:19.148404Z","iopub.execute_input":"2024-12-29T09:54:19.148692Z","iopub.status.idle":"2024-12-29T09:54:19.203080Z","shell.execute_reply.started":"2024-12-29T09:54:19.148669Z","shell.execute_reply":"2024-12-29T09:54:19.201860Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = LinearRegression()\nmodel.fit(X, y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:54:19.204154Z","iopub.execute_input":"2024-12-29T09:54:19.204501Z","iopub.status.idle":"2024-12-29T09:54:19.980708Z","shell.execute_reply.started":"2024-12-29T09:54:19.204458Z","shell.execute_reply":"2024-12-29T09:54:19.978532Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"这个值非常接近于0，意味着线性回归模型在训练数据上的表现几乎等同于一个常数模型（即预测所有目标变量的平均值）。这表明模型几乎没有解释能力，未能捕捉到feature_15及其平方项与responder_6之间的关系。","metadata":{}},{"cell_type":"code","source":"predictions = model.predict(X)\nprint(f\"R^2 Score: {r2_score(y, predictions)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:54:19.981807Z","iopub.execute_input":"2024-12-29T09:54:19.982149Z","iopub.status.idle":"2024-12-29T09:54:20.111721Z","shell.execute_reply.started":"2024-12-29T09:54:19.982120Z","shell.execute_reply":"2024-12-29T09:54:20.110432Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 尝试与其他特征做交互","metadata":{}},{"cell_type":"code","source":"# 选择数值型特征\nnumeric_features = train.select_dtypes(include=[np.number]).columns\n\n# 计算相关性矩阵\ncorrelation = train[numeric_features].corr()\n\n# 打印与feature_15和responder_6高度相关的特征\nprint(\"Correlation with feature_15:\")\nprint(correlation['feature_15'].abs().sort_values(ascending=False).head(10))\n\nprint(\"\\nCorrelation with responder_6:\")\nprint(correlation['responder_6'].abs().sort_values(ascending=False).head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:54:20.117453Z","iopub.execute_input":"2024-12-29T09:54:20.120524Z","iopub.status.idle":"2024-12-29T09:57:47.901945Z","shell.execute_reply.started":"2024-12-29T09:54:20.120469Z","shell.execute_reply":"2024-12-29T09:57:47.900431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 创建交互项\ninteraction_features = ['feature_17', 'feature_16', 'feature_29', 'feature_30']  # 根据相关性选择的特征\n\nfor feat in interaction_features:\n    interaction_name = f'feature_15_x_{feat}'\n    train[interaction_name] = train['feature_15'] * train[feat]\n\n# 打印新特征名称\nprint(\"New feature names:\", train.columns.tolist())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:57:47.903417Z","iopub.execute_input":"2024-12-29T09:57:47.903879Z","iopub.status.idle":"2024-12-29T09:57:47.990590Z","shell.execute_reply.started":"2024-12-29T09:57:47.903838Z","shell.execute_reply":"2024-12-29T09:57:47.989266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 移除可能导致数据泄露的特征\ntrain.drop(columns=['label'], inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:57:47.991810Z","iopub.execute_input":"2024-12-29T09:57:47.992182Z","iopub.status.idle":"2024-12-29T09:57:49.861561Z","shell.execute_reply.started":"2024-12-29T09:57:47.992152Z","shell.execute_reply":"2024-12-29T09:57:49.860663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:57:49.862570Z","iopub.execute_input":"2024-12-29T09:57:49.862972Z","iopub.status.idle":"2024-12-29T09:57:49.867452Z","shell.execute_reply.started":"2024-12-29T09:57:49.862944Z","shell.execute_reply":"2024-12-29T09:57:49.866287Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import r2_score\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:57:49.868632Z","iopub.execute_input":"2024-12-29T09:57:49.869128Z","iopub.status.idle":"2024-12-29T09:57:49.886705Z","shell.execute_reply.started":"2024-12-29T09:57:49.869089Z","shell.execute_reply":"2024-12-29T09:57:49.885423Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 检查缺失值\nprint(\"Missing values in X:\")\nprint(train[['feature_17', 'feature_16', 'feature_29', 'feature_30']].isnull().sum())\nprint(\"\\nMissing values in y:\")\nprint(train['responder_6'].isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:57:49.888123Z","iopub.execute_input":"2024-12-29T09:57:49.888504Z","iopub.status.idle":"2024-12-29T09:57:49.972462Z","shell.execute_reply.started":"2024-12-29T09:57:49.888459Z","shell.execute_reply":"2024-12-29T09:57:49.970988Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 填充 feature_17 中的缺失值\ntrain['feature_17'].fillna(train['feature_17'].mean(), inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:57:49.973550Z","iopub.execute_input":"2024-12-29T09:57:49.973880Z","iopub.status.idle":"2024-12-29T09:57:50.012454Z","shell.execute_reply.started":"2024-12-29T09:57:49.973851Z","shell.execute_reply":"2024-12-29T09:57:50.011249Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 再次检查缺失值以确认\nprint(\"\\nAfter filling missing values:\")\nprint(train[['feature_17', 'feature_16', 'feature_29', 'feature_30']].isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:57:50.013550Z","iopub.execute_input":"2024-12-29T09:57:50.013890Z","iopub.status.idle":"2024-12-29T09:57:50.073383Z","shell.execute_reply.started":"2024-12-29T09:57:50.013855Z","shell.execute_reply":"2024-12-29T09:57:50.072139Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 构建线性回归模型并评估R^2分数\nX = train[['feature_17', 'feature_16', 'feature_29', 'feature_30']]\ny = train['responder_6']\n\n# 将数据集分为训练集和测试集\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\nmodel = LinearRegression()\nmodel.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:57:50.074376Z","iopub.execute_input":"2024-12-29T09:57:50.074723Z","iopub.status.idle":"2024-12-29T09:57:52.030486Z","shell.execute_reply.started":"2024-12-29T09:57:50.074694Z","shell.execute_reply":"2024-12-29T09:57:52.029546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 在训练集上进行预测并评估\ntrain_predictions = model.predict(X_train)\nprint(f\"Training R^2 Score: {r2_score(y_train, train_predictions)}\")\n\n# 在测试集上进行预测并评估\ntest_predictions = model.predict(X_test)\nprint(f\"Test R^2 Score: {r2_score(y_test, test_predictions)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T09:57:52.031253Z","iopub.execute_input":"2024-12-29T09:57:52.031555Z","iopub.status.idle":"2024-12-29T09:57:52.141971Z","shell.execute_reply.started":"2024-12-29T09:57:52.031528Z","shell.execute_reply":"2024-12-29T09:57:52.140770Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"模型在训练集和测试集上的表现都非常接近于0","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}