{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-04T14:24:47.393927Z","iopub.execute_input":"2024-11-04T14:24:47.395416Z","iopub.status.idle":"2024-11-04T14:24:48.662981Z","shell.execute_reply.started":"2024-11-04T14:24:47.395343Z","shell.execute_reply":"2024-11-04T14:24:48.661654Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import TimeSeriesSplit\nfrom sklearn.metrics import r2_score\nfrom xgboost import XGBRegressor\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-11-04T14:24:48.666442Z","iopub.execute_input":"2024-11-04T14:24:48.667343Z","iopub.status.idle":"2024-11-04T14:24:50.656706Z","shell.execute_reply.started":"2024-11-04T14:24:48.667263Z","shell.execute_reply":"2024-11-04T14:24:50.655464Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 指定要加载的训练数据文件路径\ntrain_file_1 = \"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=0/part-0.parquet\"\ntrain_file_2 = \"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=1/part-0.parquet\"\n\n# 加载数据\ntrain_df_1 = pd.read_parquet(train_file_1)\ntrain_df_2 = pd.read_parquet(train_file_2)\n\n# 合并数据\ntrain_df = pd.concat([train_df_1, train_df_2], ignore_index=True)\n\nprint(\"Combined Train Data Shape:\", train_df.shape)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-04T14:24:50.658210Z","iopub.execute_input":"2024-11-04T14:24:50.658886Z","iopub.status.idle":"2024-11-04T14:24:59.047327Z","shell.execute_reply.started":"2024-11-04T14:24:50.658829Z","shell.execute_reply":"2024-11-04T14:24:59.046108Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Data Columns:\", train_df.columns)\nprint(train_df.head())\nprint(train_df.info())","metadata":{"execution":{"iopub.status.busy":"2024-11-04T14:24:59.049040Z","iopub.execute_input":"2024-11-04T14:24:59.049556Z","iopub.status.idle":"2024-11-04T14:24:59.102653Z","shell.execute_reply.started":"2024-11-04T14:24:59.049501Z","shell.execute_reply":"2024-11-04T14:24:59.101514Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_values = train_df.isnull().sum()\nprint(\"Missing Values in Each Column:\\n\", missing_values[missing_values > 0])","metadata":{"execution":{"iopub.status.busy":"2024-11-04T14:24:59.106423Z","iopub.execute_input":"2024-11-04T14:24:59.107039Z","iopub.status.idle":"2024-11-04T14:24:59.692374Z","shell.execute_reply.started":"2024-11-04T14:24:59.106996Z","shell.execute_reply":"2024-11-04T14:24:59.691071Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_values[missing_values > 0]/train_df.shape[0]","metadata":{"execution":{"iopub.status.busy":"2024-11-04T14:24:59.694069Z","iopub.execute_input":"2024-11-04T14:24:59.694563Z","iopub.status.idle":"2024-11-04T14:24:59.708598Z","shell.execute_reply.started":"2024-11-04T14:24:59.694509Z","shell.execute_reply":"2024-11-04T14:24:59.707345Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_cols = [col for col in train_df.columns if 'feature_' in col]\ncorr_matrix = train_df[feature_cols].corr()","metadata":{"execution":{"iopub.status.busy":"2024-11-04T14:24:59.710171Z","iopub.execute_input":"2024-11-04T14:24:59.710685Z","iopub.status.idle":"2024-11-04T14:26:11.842130Z","shell.execute_reply.started":"2024-11-04T14:24:59.710636Z","shell.execute_reply":"2024-11-04T14:26:11.840859Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(20, 16))\nsns.heatmap(corr_matrix, cmap='coolwarm')\nplt.title(\"Feature Correlation Matrix\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-04T14:26:11.843656Z","iopub.execute_input":"2024-11-04T14:26:11.844030Z","iopub.status.idle":"2024-11-04T14:26:13.844710Z","shell.execute_reply.started":"2024-11-04T14:26:11.843990Z","shell.execute_reply":"2024-11-04T14:26:13.843308Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.cluster.hierarchy import linkage, dendrogram\n\ndistance_matrix = 1 - np.abs(corr_matrix)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-04T14:26:13.846331Z","iopub.execute_input":"2024-11-04T14:26:13.846811Z","iopub.status.idle":"2024-11-04T14:26:13.853795Z","shell.execute_reply.started":"2024-11-04T14:26:13.846757Z","shell.execute_reply":"2024-11-04T14:26:13.852630Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"distance_matrix.min().min()","metadata":{"execution":{"iopub.status.busy":"2024-11-04T14:26:13.855412Z","iopub.execute_input":"2024-11-04T14:26:13.855973Z","iopub.status.idle":"2024-11-04T14:26:13.869467Z","shell.execute_reply.started":"2024-11-04T14:26:13.855933Z","shell.execute_reply":"2024-11-04T14:26:13.868342Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"distance_matrix.isnull().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2024-11-04T14:26:13.871115Z","iopub.execute_input":"2024-11-04T14:26:13.871672Z","iopub.status.idle":"2024-11-04T14:26:13.883219Z","shell.execute_reply.started":"2024-11-04T14:26:13.871617Z","shell.execute_reply":"2024-11-04T14:26:13.881966Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"distance_matrix = np.nan_to_num(distance_matrix, nan=1)","metadata":{"execution":{"iopub.status.busy":"2024-11-04T14:26:13.884922Z","iopub.execute_input":"2024-11-04T14:26:13.887179Z","iopub.status.idle":"2024-11-04T14:26:13.893608Z","shell.execute_reply.started":"2024-11-04T14:26:13.887116Z","shell.execute_reply":"2024-11-04T14:26:13.892284Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"linkage_matrix = linkage(distance_matrix, method='ward')","metadata":{"execution":{"iopub.status.busy":"2024-11-04T14:26:13.895152Z","iopub.execute_input":"2024-11-04T14:26:13.895619Z","iopub.status.idle":"2024-11-04T14:26:13.907543Z","shell.execute_reply.started":"2024-11-04T14:26:13.895566Z","shell.execute_reply":"2024-11-04T14:26:13.906341Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(20 ,10))\ndendrogram(linkage_matrix, labels=feature_cols, leaf_rotation=90)\nplt.title(\"Hierarchical Clustering Dendrogram\")\nplt.xlabel(\"Feature\")\nplt.ylabel(\"Distance\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-04T14:26:13.912799Z","iopub.execute_input":"2024-11-04T14:26:13.913208Z","iopub.status.idle":"2024-11-04T14:26:15.387294Z","shell.execute_reply.started":"2024-11-04T14:26:13.913167Z","shell.execute_reply":"2024-11-04T14:26:15.385946Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.cluster.hierarchy import fcluster\n\nnum_clusters = 3\ncluster_labels = fcluster(linkage_matrix, num_clusters, criterion='maxclust')\n\nfeature_clusters = pd.DataFrame({'feature': feature_cols, 'cluster': cluster_labels})\nprint(feature_clusters.head())","metadata":{"execution":{"iopub.status.busy":"2024-11-04T14:26:15.389098Z","iopub.execute_input":"2024-11-04T14:26:15.389577Z","iopub.status.idle":"2024-11-04T14:26:15.400233Z","shell.execute_reply.started":"2024-11-04T14:26:15.389512Z","shell.execute_reply":"2024-11-04T14:26:15.398753Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fcluster?","metadata":{"execution":{"iopub.status.busy":"2024-11-04T14:26:15.401992Z","iopub.execute_input":"2024-11-04T14:26:15.403051Z","iopub.status.idle":"2024-11-04T14:26:15.489551Z","shell.execute_reply.started":"2024-11-04T14:26:15.402996Z","shell.execute_reply":"2024-11-04T14:26:15.488110Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for cluster_id in range(1, num_clusters + 1):\n    print(f\"\\nAnalyzing Feature Cluster {cluster_id}\")\n    \n    cluster_features = feature_clusters[feature_clusters['cluster'] == cluster_id]['feature'].tolist()\n    \n    print(f\"Features in Cluster {cluster_id}: {cluster_features}\")\n    \n    cluster_data = train_df[cluster_features + ['responder_6']]\n    \n    print(cluster_data.describe())\n    \n    cluster_data[cluster_features].hist(figsize=(15, 10))\n    plt.suptitle(f\"Feature Distributions for Cluster {cluster_id}\")\n    plt.show()\n    \n    corr_with_target = cluster_data.corr()['responder_6'].drop('responder_6')\n    print(f\"Correlation with responder_6:\\n{corr_with_target.sort_values(ascending=False)}\")\n    \n    for feature in cluster_features:\n        plt.figure(figsize=(6,4))\n        sns.scatterplot(data=cluster_data, x=feature, y='responder_6')\n        plt.title(f\"{feature} vs responder_6\")\n        plt.show()\n    ","metadata":{"execution":{"iopub.status.busy":"2024-11-04T14:26:15.491187Z","iopub.execute_input":"2024-11-04T14:26:15.491673Z","iopub.status.idle":"2024-11-04T14:41:08.390140Z","shell.execute_reply.started":"2024-11-04T14:26:15.491620Z","shell.execute_reply":"2024-11-04T14:41:08.388649Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train_df[feature_cols]\ny = train_df['responder_6']\n\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)","metadata":{"execution":{"iopub.status.busy":"2024-11-04T14:41:08.392155Z","iopub.execute_input":"2024-11-04T14:41:08.392672Z","iopub.status.idle":"2024-11-04T14:41:20.593551Z","shell.execute_reply.started":"2024-11-04T14:41:08.392615Z","shell.execute_reply":"2024-11-04T14:41:20.591908Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tscv = TimeSeriesSplit(n_splits=5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-04T14:41:20.595759Z","iopub.execute_input":"2024-11-04T14:41:20.596607Z","iopub.status.idle":"2024-11-04T14:41:20.602428Z","shell.execute_reply.started":"2024-11-04T14:41:20.596545Z","shell.execute_reply":"2024-11-04T14:41:20.601058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 定义 XGBoost 模型参数\nxgb_params = {\n    'objective': 'reg:squarederror',\n    'n_estimators': 100,\n    'learning_rate': 0.1,\n    'max_depth': 6,\n    'random_state': 42\n}\n\nxgb_model = XGBRegressor(**xgb_params)\nr2_scores = []\n\n# 进行交叉验证\nfor train_index, test_index in tscv.split(X_scaled):\n    X_train, X_test = X_scaled[train_index], X_scaled[test_index]\n    y_train, y_test = y.iloc[train_index], y.iloc[test_index]\n    \n    xgb_model.fit(X_train, y_train)\n    y_pred = xgb_model.predict(X_test)\n    r2 = r2_score(y_test, y_pred)\n    r2_scores.append(r2)\n    print(f\"R-squared Score: {r2:.4f}\")\n\nprint(\"Average R-squared Score:\", np.mean(r2_scores))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-04T14:45:37.294865Z","iopub.execute_input":"2024-11-04T14:45:37.295672Z","iopub.status.idle":"2024-11-04T14:50:26.298506Z","shell.execute_reply.started":"2024-11-04T14:45:37.295625Z","shell.execute_reply":"2024-11-04T14:50:26.297340Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_importances = pd.Series(xgb_model.feature_importances_, index=feature_cols)\nsorted_importances = feature_importances.sort_values(ascending=False)\n\nplt.figure(figsize=(12, 20))\nsns.barplot(x=sorted_importances.values, y=sorted_importances.index)\nplt.title(\"Feature Importances from XGBoost Model\")\nplt.xlabel(\"Importance\")\nplt.ylabel(\"Feature\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-04T14:53:59.486410Z","iopub.execute_input":"2024-11-04T14:53:59.486882Z","iopub.status.idle":"2024-11-04T14:54:00.793735Z","shell.execute_reply.started":"2024-11-04T14:53:59.486835Z","shell.execute_reply":"2024-11-04T14:54:00.792346Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgb_model.feature_importances_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-04T14:55:09.226316Z","iopub.execute_input":"2024-11-04T14:55:09.226828Z","iopub.status.idle":"2024-11-04T14:55:09.237438Z","shell.execute_reply.started":"2024-11-04T14:55:09.226775Z","shell.execute_reply":"2024-11-04T14:55:09.236130Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_importances","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-04T14:55:37.060843Z","iopub.execute_input":"2024-11-04T14:55:37.061379Z","iopub.status.idle":"2024-11-04T14:55:37.073002Z","shell.execute_reply.started":"2024-11-04T14:55:37.061331Z","shell.execute_reply":"2024-11-04T14:55:37.071573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}