{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"What this notebook does: \n\n- Exploratory Data Analysis of feature distributions, types, and missingness\n\n- Correlation and redundancy checks to identify impactful and redundant features\n\n- Dimensionality reduction with PCA and t-SNE for variance explanation and visualization\n\n- Feature selection combining univariate tests, Random Forest importance, and Recursive Feature Elimination (RFE)\n\n- Clustering to detect natural groupings within the data\n\n\nI've mainly plotted for top 50 important features, you can tweak the code for the entire feature set. Hopefully, it helps. \n","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.decomposition import PCA\nfrom sklearn.manifold import TSNE\nfrom sklearn.feature_selection import SelectKBest, f_regression, RFE\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.cluster import KMeans\nfrom sklearn.preprocessing import StandardScaler\nimport warnings\nfrom statsmodels.graphics.tsaplots import plot_acf\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-23T08:23:58.063531Z","iopub.execute_input":"2025-06-23T08:23:58.063875Z","iopub.status.idle":"2025-06-23T08:24:00.861689Z","shell.execute_reply.started":"2025-06-23T08:23:58.063774Z","shell.execute_reply":"2025-06-23T08:24:00.860756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the data\nprint(\"Loading training data...\")\ntrain_df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\nprint(f\"Training data shape: {train_df.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-23T08:34:10.109810Z","iopub.execute_input":"2025-06-23T08:34:10.110600Z","iopub.status.idle":"2025-06-23T08:34:15.845465Z","shell.execute_reply.started":"2025-06-23T08:34:10.110567Z","shell.execute_reply":"2025-06-23T08:34:15.844355Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1. EXPLORATORY DATA ANALYSIS","metadata":{}},{"cell_type":"code","source":"\nprint(\"1. EXPLORATORY DATA ANALYSIS\")\n\n# Basic info about the dataset\nprint(f\"Dataset shape: {train_df.shape}\")\nprint(f\"Number of anonymized features: {len([col for col in train_df.columns if col.startswith('X')])}\")\n\n# Check for missing values\nmissing_values = train_df.isnull().sum()\nprint(f\"\\nMissing values summary:\")\nprint(f\"Total features with missing values: {(missing_values > 0).sum()}\")\nif (missing_values > 0).sum() > 0:\n    print(\"Features with most missing values:\")\n    print(missing_values[missing_values > 0].sort_values(ascending=False).head(10))\n\n# Separate known features from anonymized features\nknown_features = ['timestamp', 'bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']\nanonymized_features = [col for col in train_df.columns if col.startswith('X')]\ntarget = 'label'\n\nprint(f\"\\nKnown features: {len(known_features)}\")\nprint(f\"Anonymized features: {len(anonymized_features)}\")\n\n# Statistical summary of anonymized features\nX_features = train_df[anonymized_features]\nprint(f\"\\nAnonymized features statistical summary:\")\nprint(X_features.describe().T.head(10))\n\n# Check data types\nprint(f\"\\nData types of anonymized features:\")\nprint(X_features.dtypes.value_counts())\n\n# Distribution analysis of first few anonymized features\nfig, axes = plt.subplots(2, 3, figsize=(15, 10))\nfig.suptitle('Distribution of First 6 Anonymized Features', fontsize=16)\nfor i, feature in enumerate(anonymized_features[:6]):\n    row, col = i // 3, i % 3\n    X_features[feature].hist(bins=50, ax=axes[row, col], alpha=0.7)\n    axes[row, col].set_title(f'{feature}')\n    axes[row, col].set_xlabel('Value')\n    axes[row, col].set_ylabel('Frequency')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T10:41:14.693786Z","iopub.execute_input":"2025-05-28T10:41:14.694078Z","iopub.status.idle":"2025-05-28T10:41:36.970056Z","shell.execute_reply.started":"2025-05-28T10:41:14.694054Z","shell.execute_reply":"2025-05-28T10:41:36.968888Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. FEATURE RELATIONSHIPS ANALYSIS","metadata":{}},{"cell_type":"code","source":"print(\"\\n\" + \"=\"*50)\nprint(\"2. FEATURE RELATIONSHIPS ANALYSIS\")\nprint(\"=\"*50)\n\n# Correlation analysis for a subset of features (first 50 for visualization)\nsubset_features = anonymized_features[:50]\ncorrelation_matrix = train_df[subset_features  + [target]].corr()\n\n# Plot correlation heatmap\nplt.figure(figsize=(32, 20))\nsns.heatmap(correlation_matrix, cmap='coolwarm', center=0, \n            square=True, fmt='.2f',annot=True, cbar_kws={'shrink': 0.8})\nplt.title('Correlation Matrix - First 50 Anonymized Features + Target')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T10:41:36.970934Z","iopub.execute_input":"2025-05-28T10:41:36.971482Z","iopub.status.idle":"2025-05-28T10:41:45.936826Z","shell.execute_reply.started":"2025-05-28T10:41:36.971455Z","shell.execute_reply":"2025-05-28T10:41:45.935759Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Find highly correlated feature pairs","metadata":{}},{"cell_type":"code","source":"def find_high_correlations(corr_matrix, threshold=0.8):\n    high_corr_pairs = []\n    for i in range(len(corr_matrix.columns)):\n        for j in range(i+1, len(corr_matrix.columns)):\n            if abs(corr_matrix.iloc[i, j]) > threshold:\n                high_corr_pairs.append({\n                    'feature1': corr_matrix.columns[i],\n                    'feature2': corr_matrix.columns[j],\n                    'correlation': corr_matrix.iloc[i, j]\n                })\n    return pd.DataFrame(high_corr_pairs)\n\nhigh_corr_df = find_high_correlations(correlation_matrix, threshold=0.98)\nprint(f\"High correlation pairs (|corr| > 0.98):\")\nprint(high_corr_df.sort_values('correlation', key=abs, ascending=False).head(50))\n\n# Correlation with target variable\ntarget_correlations = train_df[anonymized_features + [target]].corr()[target].abs().sort_values(ascending=False)\nprint(f\"\\nTop 20 features most correlated with target:\")\nprint(target_correlations.head(51)[1:])  # Exclude target itself","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T10:41:45.937886Z","iopub.execute_input":"2025-05-28T10:41:45.938136Z","iopub.status.idle":"2025-05-28T10:59:06.930500Z","shell.execute_reply.started":"2025-05-28T10:41:45.938116Z","shell.execute_reply":"2025-05-28T10:59:06.928169Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. DIMENSIONALITY REDUCTION","metadata":{}},{"cell_type":"markdown","source":"# PCA Analysis","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import IncrementalPCA\nimport matplotlib.pyplot as plt\n\nprint(\"\\n\" + \"=\"*50)\nprint(\"3. DIMENSIONALITY REDUCTION\")\nprint(\"=\"*50)\n\n# 1. Use only numeric columns and limit to 200 features\nX_numeric = X_features.select_dtypes(include=[np.number]).iloc[:, :200]\n\n# 2. Fill missing values with column median\nX_clean = X_numeric.fillna(X_numeric.median())\n\n# 3. Downcast to float32 to save memory\nX_clean = X_clean.astype(np.float32)\n\n# 4. Standardize features\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X_clean)\n\n# 5. Perform Incremental PCA\nprint(\"Performing Incremental PCA...\")\nipca = IncrementalPCA(n_components=50, batch_size=200)\nX_pca = ipca.fit_transform(X_scaled)\n\n# 6. Explained variance analysis\nexplained_variance_ratio = ipca.explained_variance_ratio_\ncumsum_variance = np.cumsum(explained_variance_ratio)\nn_components_90 = np.argmax(cumsum_variance >= 0.90) + 1\nn_components_95 = np.argmax(cumsum_variance >= 0.95) + 1\n\nprint(f\"Number of components needed for 90% variance: {n_components_90}\")\nprint(f\"Number of components needed for 95% variance: {n_components_95}\")\n\n# 7. Plot explained variance\nplt.figure(figsize=(12, 5))\n\nplt.subplot(1, 2, 1)\nplt.plot(range(1, len(explained_variance_ratio) + 1),\n         explained_variance_ratio, 'bo-')\nplt.xlabel('Principal Component')\nplt.ylabel('Explained Variance Ratio')\nplt.title('PCA - Individual Explained Variance')\nplt.grid(True)\n\nplt.subplot(1, 2, 2)\nplt.plot(range(1, len(cumsum_variance) + 1),\n         cumsum_variance, 'ro-')\nplt.axhline(y=0.90, color='g', linestyle='--', label='90% variance')\nplt.axhline(y=0.95, color='b', linestyle='--', label='95% variance')\nplt.xlabel('Number of Components')\nplt.ylabel('Cumulative Explained Variance')\nplt.title('PCA - Cumulative Explained Variance')\nplt.legend()\nplt.grid(True)\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T10:59:06.933568Z","iopub.execute_input":"2025-05-28T10:59:06.933988Z","iopub.status.idle":"2025-05-28T10:59:48.943326Z","shell.execute_reply.started":"2025-05-28T10:59:06.933963Z","shell.execute_reply":"2025-05-28T10:59:48.942293Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# t-SNE visualization (on a sample for computational efficiency)","metadata":{}},{"cell_type":"code","source":"# t-SNE visualization (on a sample for computational efficiency)\nprint(\"Performing t-SNE on sample data...\")\nsample_size = min(5000, len(X_clean))\nsample_indices = np.random.choice(len(X_clean), sample_size, replace=False)\nX_sample = X_scaled[sample_indices]\ny = train_df['label']  # <-- Replace 'target' with your actual label column name\ny_sample = y.iloc[sample_indices]\n\n# Use PCA first to reduce dimensions before t-SNE\npca_pre = PCA(n_components=50)\nX_pca_sample = pca_pre.fit_transform(X_sample)\n\ntsne = TSNE(n_components=2, random_state=42, perplexity=30)\nX_tsne = tsne.fit_transform(X_pca_sample)\n\nplt.figure(figsize=(30, 8))\nscatter = plt.scatter(X_tsne[:, 0], X_tsne[:, 1], c=y_sample, alpha=0.6, cmap='viridis')\nplt.colorbar(scatter, label='Target Value')\nplt.title('t-SNE Visualization of Anonymized Features')\nplt.xlabel('t-SNE Component 1')\nplt.ylabel('t-SNE Component 2')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T10:59:48.947110Z","iopub.execute_input":"2025-05-28T10:59:48.947506Z","iopub.status.idle":"2025-05-28T11:00:19.722450Z","shell.execute_reply.started":"2025-05-28T10:59:48.947470Z","shell.execute_reply":"2025-05-28T11:00:19.721129Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. FEATURE SELECTION","metadata":{}},{"cell_type":"code","source":"from sklearn.feature_selection import SelectKBest, f_regression\nimport pandas as pd\n\nprint(\"\\n\" + \"=\"*50)\nprint(\"4. FEATURE SELECTION\")\nprint(\"=\"*50)\n\n# Univariate feature selection\nprint(\"Performing univariate feature selection...\")\nselector_univariate = SelectKBest(score_func=f_regression, k=100)\nX_selected_univariate = selector_univariate.fit_transform(X_clean, y)\n\n# Get feature scores\nanonymized_features = X_clean.columns  # <-- Make sure this matches the full feature list\nfeature_scores = pd.DataFrame({\n    'feature': anonymized_features,\n    'score': selector_univariate.scores_\n}).sort_values('score', ascending=False)\n\nprint(\"Top 20 features by univariate selection:\")\nprint(feature_scores.head(20))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:00:19.723717Z","iopub.execute_input":"2025-05-28T11:00:19.723987Z","iopub.status.idle":"2025-05-28T11:00:20.905035Z","shell.execute_reply.started":"2025-05-28T11:00:19.723957Z","shell.execute_reply":"2025-05-28T11:00:20.904246Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Random Forest feature importance","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\n\nprint(\"\\nTraining Random Forest for feature importance (with speed-up)...\")\n\nrf = RandomForestRegressor(\n    n_estimators=100,       # Reduce trees from 100 → 50 for faster training\n    max_depth=10,          # Limit depth to prevent very deep trees\n    max_features='sqrt',   # Use sqrt of features at each split (default)\n    n_jobs=-1,             # Use all CPU cores\n    random_state=42\n)\n\nrf.fit(X_clean, y)\n\nfeature_importance = pd.DataFrame({\n    'feature': anonymized_features,\n    'importance': rf.feature_importances_\n}).sort_values('importance', ascending=False)\n\nprint(\"Top 20 features by Random Forest importance:\")\nprint(feature_importance.head(20))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:00:20.905931Z","iopub.execute_input":"2025-05-28T11:00:20.906207Z","iopub.status.idle":"2025-05-28T11:05:14.628944Z","shell.execute_reply.started":"2025-05-28T11:00:20.906170Z","shell.execute_reply":"2025-05-28T11:05:14.628002Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Recursive Feature Elimination","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import Lasso\nfrom sklearn.feature_selection import RFE\n\nprint(\"\\nPerforming Recursive Feature Elimination with Lasso...\")\n\n# Initialize Lasso with a small alpha (regularization strength) and enough iterations\nlasso = Lasso(alpha=0.01, max_iter=10000, random_state=42)\n\n# RFE with step=50 (removes 50 features at a time) to speed up\nrfe = RFE(estimator=lasso, n_features_to_select=100, step=50)\n\n# Fit RFE on your data\nrfe.fit(X_clean, y)\n\n# Collect feature selection results\nrfe_features = pd.DataFrame({\n    'feature': anonymized_features,\n    'selected': rfe.support_,\n    'ranking': rfe.ranking_\n}).sort_values('ranking')\n\nselected_features_rfe = rfe_features[rfe_features['selected']]['feature'].tolist()\nprint(f\"RFE selected {len(selected_features_rfe)} features\")\nprint(\"Top 20 RFE selected features:\")\nprint(rfe_features.head(20))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:05:14.629938Z","iopub.execute_input":"2025-05-28T11:05:14.630245Z","iopub.status.idle":"2025-05-28T11:05:25.720733Z","shell.execute_reply.started":"2025-05-28T11:05:14.630183Z","shell.execute_reply":"2025-05-28T11:05:25.719288Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5. CLUSTERING ANALYSIS","metadata":{}},{"cell_type":"code","source":"print(\"\\n\" + \"=\"*50)\nprint(\"5. CLUSTERING ANALYSIS\")\nprint(\"=\"*50)\n\n# K-means clustering on PCA-reduced data\nprint(\"Performing K-means clustering...\")\nn_clusters_range = range(2, 11)\ninertias = []\n\nfor k in n_clusters_range:\n    kmeans = KMeans(n_clusters=k, random_state=42)\n    kmeans.fit(X_pca[:, :50])  # Use first 50 PCA components\n    inertias.append(kmeans.inertia_)\n\n# Plot elbow curve\nplt.figure(figsize=(10, 6))\nplt.plot(n_clusters_range, inertias, 'bo-')\nplt.xlabel('Number of Clusters')\nplt.ylabel('Inertia')\nplt.title('K-means Clustering - Elbow Method')\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:05:25.721415Z","iopub.execute_input":"2025-05-28T11:05:25.721673Z","iopub.status.idle":"2025-05-28T11:09:17.288732Z","shell.execute_reply.started":"2025-05-28T11:05:25.721652Z","shell.execute_reply":"2025-05-28T11:09:17.287891Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Analyze clusters","metadata":{}},{"cell_type":"code","source":"# Apply optimal clustering\noptimal_k = 5  # You can adjust based on elbow curve\nkmeans_final = KMeans(n_clusters=optimal_k, random_state=42)\nclusters = kmeans_final.fit_predict(X_pca[:, :50])\n\n# Analyze clusters\ncluster_analysis = pd.DataFrame({\n    'cluster': clusters,\n    'target': y\n})\n\nprint(f\"\\nCluster analysis with {optimal_k} clusters:\")\ncluster_stats = cluster_analysis.groupby('cluster')['target'].agg(['count', 'mean', 'std'])\nprint(cluster_stats)\n\n# Visualize clusters using first 2 PCA components\nplt.figure(figsize=(12, 5))\n\nplt.subplot(1, 2, 1)\nscatter = plt.scatter(X_pca[:, 0], X_pca[:, 1], c=clusters, alpha=0.6, cmap='tab10')\nplt.colorbar(scatter, label='Cluster')\nplt.xlabel('First Principal Component')\nplt.ylabel('Second Principal Component')\nplt.title('K-means Clusters in PCA Space')\n\nplt.subplot(1, 2, 2)\nscatter = plt.scatter(X_pca[:, 0], X_pca[:, 1], c=y, alpha=0.6, cmap='viridis')\nplt.colorbar(scatter, label='Target Value')\nplt.xlabel('First Principal Component')\nplt.ylabel('Second Principal Component')\nplt.title('Target Values in PCA Space')\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:09:17.289681Z","iopub.execute_input":"2025-05-28T11:09:17.289954Z","iopub.status.idle":"2025-05-28T11:10:02.340100Z","shell.execute_reply.started":"2025-05-28T11:09:17.289926Z","shell.execute_reply":"2025-05-28T11:10:02.339129Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# SUMMARY INSIGHTS","metadata":{}},{"cell_type":"code","source":"print(\"\\n\" + \"=\"*80)\nprint(\"SUMMARY INSIGHTS\")\nprint(\"=\"*80)\n\nprint(f\"1. Dataset contains {890} anonymized features\")\nprint(f\"2. {n_components_90} components explain 90% of variance, {n_components_95} explain 95%\")\nprint(f\"3. Top correlated feature with target: {target_correlations.index[1]} (corr: {target_correlations.iloc[1]:.4f})\")\nprint(f\"4. {len(high_corr_df)} feature pairs have high correlation (>0.7)\")\nprint(f\"5. Optimal number of clusters appears to be around {optimal_k}\")\n\n# Save important features for further analysis\nimportant_features = {\n    'top_univariate': feature_scores.head(50)['feature'].tolist(),\n    'top_rf_importance': feature_importance.head(50)['feature'].tolist(),\n    'rfe_selected': selected_features_rfe,\n    'high_target_corr': target_correlations.head(50).index[1:].tolist()\n}\n\ncommon_features = set(important_features['top_univariate']) & \\\n                 set(important_features['top_rf_importance']) & \\\n                 set(important_features['rfe_selected'])\nprint(f\"  Common Features selected by all 3 methods: {(common_features)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:10:02.341257Z","iopub.execute_input":"2025-05-28T11:10:02.341529Z","iopub.status.idle":"2025-05-28T11:10:02.350331Z","shell.execute_reply.started":"2025-05-28T11:10:02.341508Z","shell.execute_reply":"2025-05-28T11:10:02.349388Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"try:\n    train_df['timestamp'] = pd.to_datetime(train_df['timestamp'], unit='s')\nexcept:\n    pass  # Leave it if it's already in datetime format or anonymized\n\n# 📊 1. Distribution of the label\nplt.figure(figsize=(10, 5))\nsns.histplot(train_df['label'], kde=True, bins=100, color='steelblue')\nplt.title('Label Distribution')\nplt.xlabel('Label')\nplt.ylabel('Count')\nplt.grid(True)\nplt.show()\n\nprint(\"Label Summary Statistics:\")\nprint(train_df['label'].describe())\nprint(\"\\nNumber of Unique Labels:\", train_df['label'].nunique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-23T08:26:57.721165Z","iopub.execute_input":"2025-06-23T08:26:57.721498Z","iopub.status.idle":"2025-06-23T08:27:00.759494Z","shell.execute_reply.started":"2025-06-23T08:26:57.721475Z","shell.execute_reply":"2025-06-23T08:27:00.758345Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 📉 2. Sign Distribution (Up, Down, Neutral)\ntrain_df['label_sign'] = train_df['label'].apply(lambda x: 'up' if x > 0 else ('down' if x < 0 else 'neutral'))\n\nplt.figure(figsize=(6, 4))\nsns.countplot(data=train_df, x='label_sign', order=['up', 'neutral', 'down'], palette='Set2')\nplt.title('Label Direction Distribution')\nplt.xlabel('Movement')\nplt.ylabel('Count')\nplt.grid(True)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-23T08:27:52.794454Z","iopub.execute_input":"2025-06-23T08:27:52.794923Z","iopub.status.idle":"2025-06-23T08:27:53.256754Z","shell.execute_reply.started":"2025-06-23T08:27:52.794891Z","shell.execute_reply":"2025-06-23T08:27:53.255776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-23T08:31:08.931389Z","iopub.execute_input":"2025-06-23T08:31:08.931720Z","iopub.status.idle":"2025-06-23T08:31:08.944457Z","shell.execute_reply.started":"2025-06-23T08:31:08.931696Z","shell.execute_reply":"2025-06-23T08:31:08.943544Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['ask_qty']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-23T08:32:44.981598Z","iopub.execute_input":"2025-06-23T08:32:44.982346Z","iopub.status.idle":"2025-06-23T08:32:44.990166Z","shell.execute_reply.started":"2025-06-23T08:32:44.982319Z","shell.execute_reply":"2025-06-23T08:32:44.989157Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = train_df.sort_index()  # sorts by timestamp index\n\nrolling_mean = train_df['label'].rolling(window=1000).mean()\nrolling_std = train_df['label'].rolling(window=1000).std()\n\nplt.figure(figsize=(12, 5))\nplt.plot(rolling_mean, label='Rolling Mean (1000)', color='blue')\nplt.plot(rolling_std, label='Rolling Std (1000)', color='orange')\nplt.title('Rolling Statistics of Label Over Time')\nplt.xlabel('Time')\nplt.ylabel('Label Value')\nplt.legend()\nplt.grid(True)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-23T08:42:26.740546Z","iopub.execute_input":"2025-06-23T08:42:26.741127Z","iopub.status.idle":"2025-06-23T08:42:31.127720Z","shell.execute_reply.started":"2025-06-23T08:42:26.741092Z","shell.execute_reply":"2025-06-23T08:42:31.126563Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 🔁 4. Autocorrelation (Noise Check)\nplt.figure(figsize=(10, 4))\nplot_acf(train_df['label'].dropna(), lags=50, title=\"Autocorrelation of Label\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-23T08:43:03.351754Z","iopub.execute_input":"2025-06-23T08:43:03.352116Z","iopub.status.idle":"2025-06-23T08:44:09.912838Z","shell.execute_reply.started":"2025-06-23T08:43:03.352089Z","shell.execute_reply":"2025-06-23T08:44:09.911503Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 📊 5. Correlation with Market Features\nmarket_feats = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']\ncorr = train_df[market_feats + ['label']].corr()\n\nplt.figure(figsize=(8, 6))\nsns.heatmap(corr, annot=True, cmap='coolwarm', fmt=\".2f\")\nplt.title('Correlation with Label')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-23T08:44:19.539495Z","iopub.execute_input":"2025-06-23T08:44:19.540610Z","iopub.status.idle":"2025-06-23T08:44:19.987511Z","shell.execute_reply.started":"2025-06-23T08:44:19.540566Z","shell.execute_reply":"2025-06-23T08:44:19.986225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ⚙️ Optional: Discretize Label for Classification Experiment\ntrain_df['label_class'] = pd.qcut(train_df['label'], q=3, labels=['down', 'neutral', 'up'])\n\nplt.figure(figsize=(6, 4))\nsns.countplot(data=train_df, x='label_class', palette='viridis')\nplt.title('Label Quantile Bins')\nplt.xlabel('Label Class')\nplt.ylabel('Count')\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-23T08:44:44.971762Z","iopub.execute_input":"2025-06-23T08:44:44.972142Z","iopub.status.idle":"2025-06-23T08:44:45.233345Z","shell.execute_reply.started":"2025-06-23T08:44:44.972118Z","shell.execute_reply":"2025-06-23T08:44:45.232315Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Next steps recommendations","metadata":{}},{"cell_type":"code","source":"print(f\"\\nRECOMMENDED NEXT STEPS:\")\nprint(f\"1. Focus on the top {len(common_features)} features identified by multiple selection methods\")\nprint(f\"2. Use {n_components_90}-{n_components_95} PCA components for dimensionality reduction\")\nprint(f\"3. Consider cluster-based features as additional engineered features\")\nprint(f\"4. Remove highly correlated features to reduce redundancy\")\nprint(f\"5. Use the identified important features for model training\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:10:02.351362Z","iopub.execute_input":"2025-05-28T11:10:02.351698Z","iopub.status.idle":"2025-05-28T11:10:02.372759Z","shell.execute_reply.started":"2025-05-28T11:10:02.351669Z","shell.execute_reply":"2025-05-28T11:10:02.371826Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# TRAINING","metadata":{}},{"cell_type":"markdown","source":"Built on : https://www.kaggle.com/code/ravaghi/drw-crypto-market-prediction-ensemble","metadata":{}},{"cell_type":"code","source":"!pip install koolbox scikit-learn==1.5.2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:10:02.373959Z","iopub.execute_input":"2025-05-28T11:10:02.374358Z","iopub.status.idle":"2025-05-28T11:10:13.471033Z","shell.execute_reply.started":"2025-05-28T11:10:02.374326Z","shell.execute_reply":"2025-05-28T11:10:13.469616Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nfrom sklearn.linear_model import Ridge\nfrom lightgbm import LGBMRegressor\nfrom scipy.stats import pearsonr\nfrom xgboost import XGBRegressor\nfrom sklearn.base import clone\nfrom koolbox import Trainer\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pandas as pd\nimport numpy as np\nimport warnings\nimport optuna\nimport joblib\nimport gc\n\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:10:13.472760Z","iopub.execute_input":"2025-05-28T11:10:13.473098Z","iopub.status.idle":"2025-05-28T11:10:19.185058Z","shell.execute_reply.started":"2025-05-28T11:10:13.473067Z","shell.execute_reply":"2025-05-28T11:10:19.184245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CFG:\n    train_path = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    test_path = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    sample_sub_path = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n\n    target = \"label\"\n    n_folds = 5\n    seed = 42\n\n    run_optuna = True\n    n_optuna_trials = 250","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:10:19.186057Z","iopub.execute_input":"2025-05-28T11:10:19.186648Z","iopub.status.idle":"2025-05-28T11:10:19.191457Z","shell.execute_reply.started":"2025-05-28T11:10:19.186623Z","shell.execute_reply":"2025-05-28T11:10:19.190563Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"markdown","source":"Built on original baseline : https://www.kaggle.com/code/ravaghi/drw-crypto-market-prediction-ensemble","metadata":{}},{"cell_type":"code","source":"!pip install koolbox scikit-learn==1.5.2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:18:02.138082Z","iopub.execute_input":"2025-05-28T11:18:02.138511Z","iopub.status.idle":"2025-05-28T11:18:06.121105Z","shell.execute_reply.started":"2025-05-28T11:18:02.138483Z","shell.execute_reply":"2025-05-28T11:18:06.120007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nfrom sklearn.model_selection import KFold\nfrom sklearn.linear_model import Ridge\nfrom lightgbm import LGBMRegressor\nfrom scipy.stats import pearsonr\nfrom xgboost import XGBRegressor\nfrom sklearn.base import clone\nfrom koolbox import Trainer\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pandas as pd\nimport numpy as np\nimport warnings\nimport optuna\nimport joblib\nimport gc\n\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:18:06.122961Z","iopub.execute_input":"2025-05-28T11:18:06.123720Z","iopub.status.idle":"2025-05-28T11:18:06.130129Z","shell.execute_reply.started":"2025-05-28T11:18:06.123678Z","shell.execute_reply":"2025-05-28T11:18:06.129320Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CFG:\n    train_path = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    test_path = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    sample_sub_path = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n\n    target = \"label\"\n    n_folds = 5\n    seed = 42\n\n    run_optuna = True\n    n_optuna_trials = 250","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:18:11.536308Z","iopub.execute_input":"2025-05-28T11:18:11.536634Z","iopub.status.idle":"2025-05-28T11:18:11.541571Z","shell.execute_reply.started":"2025-05-28T11:18:11.536612Z","shell.execute_reply":"2025-05-28T11:18:11.540667Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data loading and preprocessing","metadata":{}},{"cell_type":"code","source":"def reduce_mem_usage(dataframe, dataset):    \n    print('Reducing memory usage for:', dataset)\n    initial_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    \n    for col in dataframe.columns:\n        col_type = dataframe[col].dtype\n\n        c_min = dataframe[col].min()\n        c_max = dataframe[col].max()\n        if str(col_type)[:3] == 'int':\n            if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                dataframe[col] = dataframe[col].astype(np.int8)\n            elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                dataframe[col] = dataframe[col].astype(np.int16)\n            elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                dataframe[col] = dataframe[col].astype(np.int32)\n            elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                dataframe[col] = dataframe[col].astype(np.int64)\n        else:\n            if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                dataframe[col] = dataframe[col].astype(np.float16)\n            elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                dataframe[col] = dataframe[col].astype(np.float32)\n            else:\n                dataframe[col] = dataframe[col].astype(np.float64)\n\n    final_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    print('--- Memory usage before: {:.2f} MB'.format(initial_mem_usage))\n    print('--- Memory usage after: {:.2f} MB'.format(final_mem_usage))\n    print('--- Decreased memory usage by {:.1f}%\\n'.format(100 * (initial_mem_usage - final_mem_usage) / initial_mem_usage))\n\n    return dataframe","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:18:14.746250Z","iopub.execute_input":"2025-05-28T11:18:14.747048Z","iopub.status.idle":"2025-05-28T11:18:14.756572Z","shell.execute_reply.started":"2025-05-28T11:18:14.747021Z","shell.execute_reply":"2025-05-28T11:18:14.755645Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def add_features(df):\n    data = df.copy()\n    features_df = pd.DataFrame(index=data.index)\n    \n    features_df['bid_ask_spread_proxy'] = data['ask_qty'] - data['bid_qty']\n    features_df['total_liquidity'] = data['bid_qty'] + data['ask_qty']\n    features_df['trade_imbalance'] = data['buy_qty'] - data['sell_qty']\n    features_df['total_trades'] = data['buy_qty'] + data['sell_qty']\n    \n    features_df['volume_per_trade'] = data['volume'] / (data['buy_qty'] + data['sell_qty'] + 1e-8)\n    features_df['buy_volume_ratio'] = data['buy_qty'] / (data['volume'] + 1e-8)\n    features_df['sell_volume_ratio'] = data['sell_qty'] / (data['volume'] + 1e-8)\n    \n    features_df['buying_pressure'] = data['buy_qty'] / (data['buy_qty'] + data['sell_qty'] + 1e-8)\n    features_df['selling_pressure'] = data['sell_qty'] / (data['buy_qty'] + data['sell_qty'] + 1e-8)\n    \n    features_df['order_imbalance'] = (data['bid_qty'] - data['ask_qty']) / (data['bid_qty'] + data['ask_qty'] + 1e-8)\n    features_df['order_imbalance_abs'] = np.abs(features_df['order_imbalance'])\n    features_df['bid_liquidity_ratio'] = data['bid_qty'] / (data['volume'] + 1e-8)\n    features_df['ask_liquidity_ratio'] = data['ask_qty'] / (data['volume'] + 1e-8)\n    features_df['market_depth'] = data['bid_qty'] + data['ask_qty']\n    features_df['depth_imbalance'] = features_df['market_depth'] - data['volume']\n    \n    features_df['buy_sell_ratio'] = data['buy_qty'] / (data['sell_qty'] + 1e-8)\n    features_df['bid_ask_ratio'] = data['bid_qty'] / (data['ask_qty'] + 1e-8)\n    features_df['volume_liquidity_ratio'] = data['volume'] / (data['bid_qty'] + data['ask_qty'] + 1e-8)\n\n    features_df['buy_volume_product'] = data['buy_qty'] * data['volume']\n    features_df['sell_volume_product'] = data['sell_qty'] * data['volume']\n    features_df['bid_ask_product'] = data['bid_qty'] * data['ask_qty']\n    \n    features_df['market_competition'] = (data['buy_qty'] * data['sell_qty']) / ((data['buy_qty'] + data['sell_qty']) + 1e-8)\n    features_df['liquidity_competition'] = (data['bid_qty'] * data['ask_qty']) / ((data['bid_qty'] + data['ask_qty']) + 1e-8)\n    \n    total_activity = data['buy_qty'] + data['sell_qty'] + data['bid_qty'] + data['ask_qty']\n    features_df['market_activity'] = total_activity\n    features_df['activity_concentration'] = data['volume'] / (total_activity + 1e-8)\n    \n    features_df['info_arrival_rate'] = (data['buy_qty'] + data['sell_qty']) / (data['volume'] + 1e-8)\n    features_df['market_making_intensity'] = (data['bid_qty'] + data['ask_qty']) / (data['buy_qty'] + data['sell_qty'] + 1e-8)\n    features_df['effective_spread_proxy'] = np.abs(data['buy_qty'] - data['sell_qty']) / (data['volume'] + 1e-8)\n    \n    lambda_decay = 0.95\n    ofi = data['buy_qty'] - data['sell_qty']\n    features_df['order_flow_imbalance_ewm'] = ofi.ewm(alpha=1-lambda_decay).mean()\n\n    features_df = features_df.replace([np.inf, -np.inf], np.nan)\n    \n    return features_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:18:18.032455Z","iopub.execute_input":"2025-05-28T11:18:18.032803Z","iopub.status.idle":"2025-05-28T11:18:18.045371Z","shell.execute_reply.started":"2025-05-28T11:18:18.032778Z","shell.execute_reply":"2025-05-28T11:18:18.044282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cols_to_drop = [\n    'X697', 'X698', 'X699', 'X700', 'X701', 'X702', 'X703', 'X704', 'X705', 'X706', \n    'X707', 'X708', 'X709', 'X710', 'X711', 'X712', 'X713', 'X714', 'X715', 'X716',\n    'X717', 'X864', 'X867', 'X869', 'X870', 'X871', 'X872', 'X104', 'X110', 'X116',\n    'X122', 'X128', 'X134', 'X146', 'X152', 'X158', 'X164', 'X170', 'X176',\n    'X182', 'X351', 'X357', 'X363', 'X369', 'X375', 'X381', 'X387', 'X393', 'X399',\n    'X405', 'X411', 'X417', 'X423', 'X429', 'X46',  'X50', 'X45', 'X49', 'X40',\n    'X44', 'X39', 'X43', 'X6', 'X8', 'X34', 'X38', 'X35', 'X16', 'X1','X14' \n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:18:20.758880Z","iopub.execute_input":"2025-05-28T11:18:20.759237Z","iopub.status.idle":"2025-05-28T11:18:20.764896Z","shell.execute_reply.started":"2025-05-28T11:18:20.759178Z","shell.execute_reply":"2025-05-28T11:18:20.763997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_parquet(CFG.train_path).reset_index(drop=True)\ntest = pd.read_parquet(CFG.test_path).reset_index(drop=True)\n\ntrain = train.drop(columns=cols_to_drop)\ntest = test.drop(columns=[\"label\"] + cols_to_drop)\n\ntrain = reduce_mem_usage(train, \"train\")\ntest = reduce_mem_usage(test, \"test\")\n\nX = train.drop(CFG.target, axis=1)\ny = train[CFG.target]\nX_test = test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:18:22.695328Z","iopub.execute_input":"2025-05-28T11:18:22.695624Z","execution_failed":"2025-05-28T11:19:21.671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = pd.concat([add_features(X), X], axis=1)\nX_test = pd.concat([add_features(X_test), X_test], axis=1)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-05-28T11:19:21.671Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training base models","metadata":{}},{"cell_type":"code","source":"def _pearsonr(y_true, y_pred):\n    return pearsonr(y_true, y_pred)[0]","metadata":{"trusted":true,"execution":{"execution_failed":"2025-05-28T11:19:21.672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_params = {\n    \"boosting_type\": \"gbdt\",\n    \"colsample_bytree\": 0.5625888953382505,\n    \"learning_rate\": 0.029312951475451557,\n    \"min_child_samples\": 63,\n    \"min_child_weight\": 0.11456572852335424,\n    \"n_estimators\": 126,\n    \"n_jobs\": -1,\n    \"num_leaves\": 37,\n    \"random_state\": 42,\n    \"reg_alpha\": 85.2476527854083,\n    \"reg_lambda\": 99.38305361388907,\n    \"subsample\": 0.450669817684892,\n    \"verbose\": -1\n}\n\nlgbm_goss_params = {\n    \"boosting_type\": \"goss\",\n    \"colsample_bytree\": 0.34695458228489784,\n    \"learning_rate\": 0.031023014900595287,\n    \"min_child_samples\": 30,\n    \"min_child_weight\": 0.4727729225033618,\n    \"n_estimators\": 220,\n    \"n_jobs\": -1,\n    \"num_leaves\": 58,\n    \"random_state\": 42,\n    \"reg_alpha\": 38.665994901468224,\n    \"reg_lambda\": 92.76991677464294,\n    \"subsample\": 0.4810891284493255,\n    \"verbose\": -1\n}\n\nxgb_params = {\n    \"colsample_bylevel\": 0.4778015829774066,\n    \"colsample_bynode\": 0.362764358742407,\n    \"colsample_bytree\": 0.7107423488010493,\n    \"gamma\": 1.7094857725240398,\n    \"learning_rate\": 0.02213323588455387,\n    \"max_depth\": 20,\n    \"max_leaves\": 12,\n    \"min_child_weight\": 16,\n    \"n_estimators\": 1667,\n    \"n_jobs\": -1,\n    \"random_state\": 42,\n    \"reg_alpha\": 39.352415706891264,\n    \"reg_lambda\": 75.44843704068275,\n    \"subsample\": 0.06566669853471274,\n    \"verbosity\": 0\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:10:19.231359Z","iopub.status.idle":"2025-05-28T11:10:19.231713Z","shell.execute_reply.started":"2025-05-28T11:10:19.231526Z","shell.execute_reply":"2025-05-28T11:10:19.231541Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scores = {}\noof_preds = {}\ntest_preds = {}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:10:19.233413Z","iopub.status.idle":"2025-05-28T11:10:19.233723Z","shell.execute_reply.started":"2025-05-28T11:10:19.233557Z","shell.execute_reply":"2025-05-28T11:10:19.233573Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# LightGBM (gbdt)","metadata":{}},{"cell_type":"code","source":"\nlgbm_trainer = Trainer(\n    LGBMRegressor(**lgbm_params),\n    cv=KFold(n_splits=5, shuffle=False),\n    metric=_pearsonr,\n    task=\"regression\",\n    metric_precision=6\n)\n\nlgbm_trainer.fit(X, y)\n\nscores[\"LightGBM (gbdt)\"] = lgbm_trainer.fold_scores\noof_preds[\"LightGBM (gbdt)\"] = lgbm_trainer.oof_preds\ntest_preds[\"LightGBM (gbdt)\"] = lgbm_trainer.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:10:19.235134Z","iopub.status.idle":"2025-05-28T11:10:19.235461Z","shell.execute_reply.started":"2025-05-28T11:10:19.235335Z","shell.execute_reply":"2025-05-28T11:10:19.235351Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# LightGBM (goss)","metadata":{}},{"cell_type":"code","source":"lgbm_goss_trainer = Trainer(\n    LGBMRegressor(**lgbm_goss_params),\n    cv=KFold(n_splits=5, shuffle=False),\n    metric=_pearsonr,\n    task=\"regression\",\n    metric_precision=6\n)\n\nlgbm_goss_trainer.fit(X, y)\n\nscores[\"LightGBM (goss)\"] = lgbm_goss_trainer.fold_scores\noof_preds[\"LightGBM (goss)\"] = lgbm_goss_trainer.oof_preds\ntest_preds[\"LightGBM (goss)\"] = lgbm_goss_trainer.predict(X_test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:10:19.236281Z","iopub.status.idle":"2025-05-28T11:10:19.236552Z","shell.execute_reply.started":"2025-05-28T11:10:19.236422Z","shell.execute_reply":"2025-05-28T11:10:19.236436Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# XGBoost","metadata":{}},{"cell_type":"code","source":"xgb_trainer = Trainer(\n    XGBRegressor(**xgb_params),\n    cv=KFold(n_splits=5, shuffle=False),\n    metric=_pearsonr,\n    task=\"regression\",\n    metric_precision=6\n)\n\nxgb_trainer.fit(X, y)\n\nscores[\"XGBoost\"] = xgb_trainer.fold_scores\noof_preds[\"XGBoost\"] = xgb_trainer.oof_preds\ntest_preds[\"XGBoost\"] = xgb_trainer.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:10:19.237836Z","iopub.status.idle":"2025-05-28T11:10:19.238119Z","shell.execute_reply.started":"2025-05-28T11:10:19.237970Z","shell.execute_reply":"2025-05-28T11:10:19.237981Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Ensembling with Ridge","metadata":{}},{"cell_type":"code","source":"def plot_weights(weights, title):\n    sorted_indices = np.argsort(weights[0])[::-1]\n    sorted_coeffs = np.array(weights[0])[sorted_indices]\n    sorted_model_names = np.array(list(oof_preds.keys()))[sorted_indices]\n\n    plt.figure(figsize=(10, weights.shape[1] * 0.5))\n    ax = sns.barplot(x=sorted_coeffs, y=sorted_model_names, palette=\"RdYlGn_r\")\n\n    for i, (value, name) in enumerate(zip(sorted_coeffs, sorted_model_names)):\n        if value >= 0:\n            ax.text(value, i, f\"{value:.3f}\", va=\"center\", ha=\"left\", color=\"black\")\n        else:\n            ax.text(value, i, f\"{value:.3f}\", va=\"center\", ha=\"right\", color=\"black\")\n\n    xlim = ax.get_xlim()\n    ax.set_xlim(xlim[0] - 0.1 * abs(xlim[0]), xlim[1] + 0.1 * abs(xlim[1]))\n\n    plt.title(title)\n    plt.xlabel(\"\")\n    plt.ylabel(\"\")\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:10:19.239219Z","iopub.status.idle":"2025-05-28T11:10:19.239496Z","shell.execute_reply.started":"2025-05-28T11:10:19.239368Z","shell.execute_reply":"2025-05-28T11:10:19.239383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = pd.DataFrame(oof_preds)\nX_test = pd.DataFrame(test_preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:10:19.240345Z","iopub.status.idle":"2025-05-28T11:10:19.240581Z","shell.execute_reply.started":"2025-05-28T11:10:19.240461Z","shell.execute_reply":"2025-05-28T11:10:19.240471Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"joblib.dump(X, \"oof_preds.pkl\")\njoblib.dump(X_test, \"test_preds.pkl\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:10:19.242469Z","iopub.status.idle":"2025-05-28T11:10:19.242751Z","shell.execute_reply.started":"2025-05-28T11:10:19.242636Z","shell.execute_reply":"2025-05-28T11:10:19.242647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def objective(trial):    \n    params = {\n        \"random_state\": CFG.seed,\n        \"alpha\": trial.suggest_float(\"alpha\", 0, 1000),\n        \"tol\": trial.suggest_float(\"tol\", 1e-6, 1e-2),\n        \"fit_intercept\": trial.suggest_categorical(\"fit_intercept\", [True, False]),\n        \"positive\": trial.suggest_categorical(\"positive\", [True, False])\n    }\n\n    trainer = Trainer(\n        Ridge(**params),\n        cv=KFold(n_splits=5, shuffle=False),\n        metric=_pearsonr,\n        task=\"regression\",\n        verbose=False\n    )\n    trainer.fit(X, y)\n    \n    return np.mean(trainer.fold_scores)\n\nif CFG.run_optuna:\n    sampler = optuna.samplers.TPESampler(seed=CFG.seed, multivariate=True)\n    study = optuna.create_study(direction=\"maximize\", sampler=sampler)\n    study.optimize(objective, n_trials=CFG.n_optuna_trials, n_jobs=-1, catch=(ValueError,))\n    best_params = study.best_params\n\n    ridge_params = {\n        \"random_state\": CFG.seed,\n        \"alpha\": best_params[\"alpha\"],\n        \"tol\": best_params[\"tol\"],\n        \"fit_intercept\": best_params[\"fit_intercept\"],\n        \"positive\": best_params[\"positive\"]\n    }\nelse:\n    ridge_params = {\n        \"random_state\": CFG.seed\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:10:19.243971Z","iopub.status.idle":"2025-05-28T11:10:19.244314Z","shell.execute_reply.started":"2025-05-28T11:10:19.244139Z","shell.execute_reply":"2025-05-28T11:10:19.244154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ridge_trainer = Trainer(\n    Ridge(**ridge_params),\n    cv=KFold(n_splits=5, shuffle=False),\n    metric=_pearsonr,\n    task=\"regression\",\n    metric_precision=6\n)\n\nridge_trainer.fit(X, y)\n\nscores[\"Ridge (ensemble)\"] = ridge_trainer.fold_scores\nridge_test_preds = ridge_trainer.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:10:19.245558Z","iopub.status.idle":"2025-05-28T11:10:19.245854Z","shell.execute_reply.started":"2025-05-28T11:10:19.245727Z","shell.execute_reply":"2025-05-28T11:10:19.245741Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ridge_coeffs = np.zeros((1, X.shape[1]))\nfor m in ridge_trainer.estimators:\n    ridge_coeffs += m.coef_\nridge_coeffs = ridge_coeffs / len(ridge_trainer.estimators)\n\nplot_weights(ridge_coeffs, \"Ridge Coefficients\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:10:19.248740Z","iopub.status.idle":"2025-05-28T11:10:19.249006Z","shell.execute_reply.started":"2025-05-28T11:10:19.248886Z","shell.execute_reply":"2025-05-28T11:10:19.248898Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"sub = pd.read_csv(CFG.sample_sub_path)\nsub[\"prediction\"] = ridge_test_preds\nsub.to_csv(f\"sub_ridge_{np.mean(scores['Ridge (ensemble)']):.6f}.csv\", index=False)\nsub.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:10:19.249961Z","iopub.status.idle":"2025-05-28T11:10:19.250299Z","shell.execute_reply.started":"2025-05-28T11:10:19.250120Z","shell.execute_reply":"2025-05-28T11:10:19.250135Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Results","metadata":{}},{"cell_type":"code","source":"scores = pd.DataFrame(scores)\nmean_scores = scores.mean().sort_values(ascending=False)\norder = scores.mean().sort_values(ascending=False).index.tolist()\n\nmin_score = mean_scores.min()\nmax_score = mean_scores.max()\npadding = (max_score - min_score) * 0.5\nlower_limit = min_score - padding\nupper_limit = max_score + padding\n\nfig, axs = plt.subplots(1, 2, figsize=(15, scores.shape[1] * 0.5))\n\nboxplot = sns.boxplot(data=scores, order=order, ax=axs[0], orient=\"h\", color=\"grey\")\naxs[0].set_title(f\"Fold Score\")\naxs[0].set_xlabel(\"\")\naxs[0].set_ylabel(\"\")\n\nbarplot = sns.barplot(x=mean_scores.values, y=mean_scores.index, ax=axs[1], color=\"grey\")\naxs[1].set_title(f\"Average Score\")\naxs[1].set_xlabel(\"\")\naxs[1].set_xlim(left=lower_limit, right=upper_limit)\naxs[1].set_ylabel(\"\")\n\nfor i, (score, model) in enumerate(zip(mean_scores.values, mean_scores.index)):\n    color = \"cyan\" if \"ensemble\" in model.lower() else \"grey\"\n    barplot.patches[i].set_facecolor(color)\n    boxplot.patches[i].set_facecolor(color)\n    barplot.text(score, i, round(score, 6), va=\"center\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T11:10:19.251356Z","iopub.status.idle":"2025-05-28T11:10:19.251638Z","shell.execute_reply.started":"2025-05-28T11:10:19.251483Z","shell.execute_reply":"2025-05-28T11:10:19.251496Z"}},"outputs":[],"execution_count":null}]}