{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Comprehensive Crypto Market Prediction - Deep Dive EDA\n\n##  Introduction\n\nThis notebook provides an in-depth exploratory data analysis (EDA) for the DRW Crypto Market Prediction competition. We'll systematically analyze the dataset to understand patterns, relationships, and insights that will guide our modeling approach.\n\n\n---","metadata":{}},{"cell_type":"markdown","source":"## 1. Import Libraries and Setup","metadata":{}},{"cell_type":"code","source":"# Core data manipulation and analysis\nimport pandas as pd\nimport numpy as np\nimport warnings\nfrom typing import Tuple, List, Dict, Any\n\n# Visualization libraries\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\n\n# Statistical analysis\nfrom scipy import stats\nfrom scipy.stats import pearsonr, spearmanr, normaltest, jarque_bera\nfrom sklearn.preprocessing import StandardScaler, RobustScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.manifold import TSNE\n\n# Set display options\npd.set_option('display.max_columns', None)\npd.set_option('display.width', None)\nwarnings.filterwarnings('ignore')\n\n# Plotting style\nplt.style.use('seaborn-v0_8')\nsns.set_palette(\"husl\")\n\nprint(\"📦 All libraries imported successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T20:20:22.870218Z","iopub.execute_input":"2025-05-28T20:20:22.870399Z","iopub.status.idle":"2025-05-28T20:20:26.062360Z","shell.execute_reply.started":"2025-05-28T20:20:22.870380Z","shell.execute_reply":"2025-05-28T20:20:26.061694Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Data Loading","metadata":{}},{"cell_type":"code","source":"train = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\ntest = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\nsample_sub = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\n\nprint(f\"Train shape: {train.shape}\")\nprint(f\"Test shape: {test.shape}\")\nprint(f\"Sample submission shape: {sample_sub.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T20:20:26.063662Z","iopub.execute_input":"2025-05-28T20:20:26.064047Z","iopub.status.idle":"2025-05-28T20:21:18.048114Z","shell.execute_reply.started":"2025-05-28T20:20:26.064030Z","shell.execute_reply":"2025-05-28T20:21:18.047016Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  3. Basic Data Overview","metadata":{}},{"cell_type":"code","source":"\nprint(\"=== BASIC DATA OVERVIEW ===\")\nprint(f\"Training columns: {train.shape[1]}\")\nprint(f\"Test columns: {test.shape[1]}\")\n\n# Check common columns\ntrain_cols = set(train.columns)\ntest_cols = set(test.columns)\ncommon_cols = train_cols.intersection(test_cols)\ntrain_only = train_cols - test_cols\ntest_only = test_cols - train_cols\n\nprint(f\"Common columns: {len(common_cols)}\")\nprint(f\"Train only: {list(train_only)}\")\nprint(f\"Test only: {list(test_only)}\")\n\n# Data types\nprint(f\"\\nData types:\")\nprint(train.dtypes.value_counts())\n\n# Memory usage\ntrain_memory = train.memory_usage(deep=True).sum() / 1024**2\ntest_memory = test.memory_usage(deep=True).sum() / 1024**2\nprint(f\"\\nMemory usage:\")\nprint(f\"Train: {train_memory:.1f} MB\")\nprint(f\"Test: {test_memory:.1f} MB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T20:21:18.049268Z","iopub.execute_input":"2025-05-28T20:21:18.049566Z","iopub.status.idle":"2025-05-28T20:21:18.149417Z","shell.execute_reply.started":"2025-05-28T20:21:18.049539Z","shell.execute_reply":"2025-05-28T20:21:18.148563Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Target Variable Analysis\n","metadata":{}},{"cell_type":"code","source":"target = train['label']\n\nprint(\"=== TARGET VARIABLE ANALYSIS ===\")\nprint(f\"Count: {len(target):,}\")\nprint(f\"Mean: {target.mean():.6f}\")\nprint(f\"Median: {target.median():.6f}\")\nprint(f\"Std: {target.std():.6f}\")\nprint(f\"Min: {target.min():.6f}\")\nprint(f\"Max: {target.max():.6f}\")\nprint(f\"Skewness: {target.skew():.6f}\")\nprint(f\"Kurtosis: {target.kurtosis():.6f}\")\n\n# Normality test\ntry:\n    stat, p_value = normaltest(target.dropna())\n    print(f\"Normality test p-value: {p_value:.6f}\")\n    if p_value < 0.05:\n        print(\"Target is NOT normally distributed\")\n    else:\n        print(\"Target appears normally distributed\")\nexcept:\n    print(\"Could not perform normality test\")\n\n# Outliers using IQR\nQ1, Q3 = target.quantile(0.25), target.quantile(0.75)\nIQR = Q3 - Q1\nlower_bound = Q1 - 1.5 * IQR\nupper_bound = Q3 + 1.5 * IQR\noutliers = target[(target < lower_bound) | (target > upper_bound)]\nprint(f\"Outliers (IQR method): {len(outliers)} ({len(outliers)/len(target)*100:.2f}%)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T20:21:18.150149Z","iopub.execute_input":"2025-05-28T20:21:18.150324Z","iopub.status.idle":"2025-05-28T20:21:18.256767Z","shell.execute_reply.started":"2025-05-28T20:21:18.150310Z","shell.execute_reply":"2025-05-28T20:21:18.256114Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. Target Variable Visualization","metadata":{}},{"cell_type":"code","source":"\nfig, axes = plt.subplots(2, 2, figsize=(16, 12), constrained_layout=True)\nfig.suptitle('Target Variable Analysis', fontsize=18, y=1.02, fontweight='bold')\n\n# Histogram with KDE \nsns.histplot(target, bins=50, kde=True, ax=axes[0,0], color='skyblue', edgecolor='navy', alpha=0.7)\naxes[0,0].axvline(target.mean(), color='red', linestyle='--', linewidth=2, label=f'Mean: {target.mean():.4f}')\naxes[0,0].axvline(target.median(), color='green', linestyle='--', linewidth=2, label=f'Median: {target.median():.4f}')\naxes[0,0].set_title('Distribution Histogram', fontsize=14, pad=10)\naxes[0,0].set_xlabel('Target Value', fontsize=12)\naxes[0,0].set_ylabel('Frequency', fontsize=12)\naxes[0,0].legend(fontsize=10, framealpha=0.9)\naxes[0,0].grid(True, alpha=0.3)\n# Box plot with annotations\nbox = axes[0,1].boxplot(target, patch_artist=True, \n                        boxprops=dict(facecolor='lightgreen', alpha=0.7),\n                        whiskerprops=dict(color='green', linewidth=1.5),\n                        capprops=dict(color='green', linewidth=1.5),\n                        medianprops=dict(color='darkred', linewidth=2))\n\n# summary statistics\nstats_text = f\"\"\"\nMin: {np.min(target):.2f}\nQ1: {np.percentile(target, 25):.2f}\nMedian: {np.median(target):.2f}\nQ3: {np.percentile(target, 75):.2f}\nMax: {np.max(target):.2f}\nIQR: {np.percentile(target, 75) - np.percentile(target, 25):.2f}\n\"\"\"\naxes[0,1].text(1.2, 0.5, stats_text, transform=axes[0,1].transAxes, \n              bbox=dict(facecolor='white', alpha=0.8), fontsize=10)\naxes[0,1].set_title('Distribution Box Plot', fontsize=14, pad=10)\naxes[0,1].set_ylabel('Target Value', fontsize=12)\naxes[0,1].grid(True, alpha=0.3)\n\n# Time series with rolling average\nsample_size = min(1000, len(target))\nrolling_window = sample_size // 20  # 5% of sample size\n\naxes[1,0].plot(target[:sample_size], alpha=0.5, color='purple', label='Raw data')\naxes[1,0].plot(pd.Series(target[:sample_size]).rolling(rolling_window).mean(), \n              color='darkorange', linewidth=2, label=f'Rolling mean (window={rolling_window})')\naxes[1,0].set_title(f'Time Series (First {sample_size} points)', fontsize=14, pad=10)\naxes[1,0].set_xlabel('Index', fontsize=12)\naxes[1,0].set_ylabel('Target Value', fontsize=12)\naxes[1,0].legend(fontsize=10)\naxes[1,0].grid(True, alpha=0.3)\n\n# Q-Q plot with R² value\nstats.probplot(target, dist=\"norm\", plot=axes[1,1])\naxes[1,1].lines[0].set_markerfacecolor('blue')\naxes[1,1].lines[0].set_markersize(4.0)\naxes[1,1].lines[1].set_color('red')\naxes[1,1].lines[1].set_linewidth(2.0)\n\n# Calculate R² for the Q-Q plot\n(osm, osr), (slope, intercept, r) = stats.probplot(target, dist=\"norm\")\naxes[1,1].text(0.05, 0.9, f'R² = {r**2:.3f}', transform=axes[1,1].transAxes,\n              bbox=dict(facecolor='white', alpha=0.8))\naxes[1,1].set_title('Normality Q-Q Plot', fontsize=14, pad=10)\naxes[1,1].grid(True, alpha=0.3)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T20:22:42.998450Z","iopub.execute_input":"2025-05-28T20:22:42.998718Z","iopub.status.idle":"2025-05-28T20:22:46.472984Z","shell.execute_reply.started":"2025-05-28T20:22:42.998700Z","shell.execute_reply":"2025-05-28T20:22:46.472167Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 1. Distribution Histogram\n**Shape**: The distribution appears roughly symmetric with a concentration around zero\n\n**Spread**: The mean (0.0361) and median (0.0183) are very close, confirming symmetry\n\n**Spread**: Values range from approximately -20 to +20\n\n*Notable Features:*\n\nThe distribution shows distinct peaks at regular intervals (every 10 units)\n\nThis \"comb-like\" pattern suggests discrete measurement or quantization in the data collection\n\n---\n\n","metadata":{}},{"cell_type":"markdown","source":"### 2. Box Plot\n**Central Spread**: The interquartile range (IQR) appears relatively narrow compared to the full range\n\n**Outliers**: Multiple points extend beyond the whiskers at both extremes (-20 and +20)\n\n**Symmetry**: The median is centered in the box, and whisker lengths are roughly equal, confirming the symmetry seen in the histogram\n\n**Potential Issues**: The extreme values may represent:\n\n*True outliers*\n\n*Measurement limits (clipping at ±20)*\n\n*Artificial boundaries in the data collection*\n\n\n---","metadata":{}},{"cell_type":"markdown","source":"### 3. Time Series Plot (First 1000 Points)\n**Pattern**: Shows clear, regular oscillations between positive and negative values\n\n**Amplitude**: The oscillations reach the full range (±20) seen in other plots\n\n**Frequency**: The pattern appears to have consistent periodicity\n\nImplications:\n\n*Strong autocorrelation likely exists*\n\n*May represent seasonal patterns or system oscillations*\n\n*Could indicate measurement artifacts if the pattern is too perfect*\n\n---\n\n","metadata":{}},{"cell_type":"markdown","source":"### 4. Q-Q Plot (Normality Test)\n**Normality Assessment**: Points mostly follow the reference line but deviate at both extremes\n\n**Deviations**:\n\n*Tails are heavier than normal (points curve above the line at high values and below at low values)*\n\n*The discrete nature of the data creates visible \"steps\" in the plot*\n\n*Goodness of Fit: The R² value (0.862) is moderately high but imperfect due to the discrete nature (increments of 10, like -20, -10, 0, 10, 20)*\n\n---","metadata":{}},{"cell_type":"markdown","source":"### Key Findings:\n* **Discrete Nature**: The data appears quantized (recorded in discrete increments), which affects all analyses\n\n* **Potential Boundedness**: The ±20 limits may represent measurement boundaries rather than true data limits\n\n* **Strong Periodicity**: The time series shows remarkably regular oscillations worth investigating\n\n* **Normality**: While roughly symmetric, the distribution isn't perfectly normal, especially in the tails","metadata":{}},{"cell_type":"markdown","source":"##  6. Missing Values check \n","metadata":{}},{"cell_type":"code","source":"print(\"=== MISSING VALUES ANALYSIS ===\")\n\n# Training data\ntrain_missing = train.isnull().sum()\ntrain_missing_pct = (train_missing / len(train)) * 100\ntrain_missing_df = pd.DataFrame({\n    'Missing_Count': train_missing,\n    'Missing_Percent': train_missing_pct\n})\ntrain_missing_df = train_missing_df[train_missing_df['Missing_Count'] > 0].sort_values('Missing_Count', ascending=False)\n\nprint(f\"Training data:\")\nprint(f\"Columns with missing values: {len(train_missing_df)}\")\nif len(train_missing_df) > 0:\n    print(\"Top 10 columns with missing values:\")\n    print(train_missing_df.head(10))\nelse:\n    print(\"✅ No missing values in training data!\")\n\n# Test data\ntest_missing = test.isnull().sum()\ntest_missing_pct = (test_missing / len(test)) * 100\ntest_missing_df = pd.DataFrame({\n    'Missing_Count': test_missing,\n    'Missing_Percent': test_missing_pct\n})\ntest_missing_df = test_missing_df[test_missing_df['Missing_Count'] > 0].sort_values('Missing_Count', ascending=False)\n\nprint(f\"\\nTest data:\")\nprint(f\"Columns with missing values: {len(test_missing_df)}\")\nif len(test_missing_df) > 0:\n    print(\"Top 10 columns with missing values:\")\n    print(test_missing_df.head(10))\nelse:\n    print(\"✅ No missing values in test data!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T20:21:19.883073Z","iopub.execute_input":"2025-05-28T20:21:19.883268Z","iopub.status.idle":"2025-05-28T20:21:22.971651Z","shell.execute_reply.started":"2025-05-28T20:21:19.883252Z","shell.execute_reply":"2025-05-28T20:21:22.970668Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  7. Feature Correlation Analysis","metadata":{}},{"cell_type":"code","source":"print(\"=== FEATURE CORRELATION ANALYSIS ===\")\n\n# Get feature columns (exclude target)\nfeature_cols = [col for col in train.columns if col != 'label']\nprint(f\"Analyzing {len(feature_cols)} features\")\n\n# Calculate correlations with target\ncorrelations = []\ntarget = train['label']\n\nfor col in feature_cols:\n    try:\n        corr, p_value = pearsonr(train[col].fillna(0), target)\n        if not np.isnan(corr):\n            correlations.append({\n                'Feature': col,\n                'Correlation': corr,\n                'Abs_Correlation': abs(corr),\n                'P_Value': p_value\n            })\n    except:\n        continue\n\n# Create correlation dataframe\ncorr_df = pd.DataFrame(correlations)\ncorr_df = corr_df.sort_values('Abs_Correlation', ascending=False)\n\nprint(f\"Successfully calculated correlations for {len(corr_df)} features\")\nprint(f\"\\nTop 20 features by absolute correlation:\")\nprint(\"-\" * 60)\nfor i, row in corr_df.head(20).iterrows():\n    significance = \"***\" if row['P_Value'] < 0.001 else \"**\" if row['P_Value'] < 0.01 else \"*\" if row['P_Value'] < 0.05 else \"\"\n    print(f\"{row['Feature']:>15}: {row['Correlation']:>8.6f} {significance}\")\n\n# Correlation strength distribution\nstrong_corr = corr_df[corr_df['Abs_Correlation'] > 0.1]\nmoderate_corr = corr_df[(corr_df['Abs_Correlation'] > 0.05) & (corr_df['Abs_Correlation'] <= 0.1)]\nweak_corr = corr_df[corr_df['Abs_Correlation'] <= 0.05]\n\nprint(f\"\\nCorrelation strength distribution:\")\nprint(f\"Strong (|r| > 0.1): {len(strong_corr)} features\")\nprint(f\"Moderate (0.05 < |r| ≤ 0.1): {len(moderate_corr)} features\")\nprint(f\"Weak (|r| ≤ 0.05): {len(weak_corr)} features\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T20:21:22.974076Z","iopub.execute_input":"2025-05-28T20:21:22.974317Z","iopub.status.idle":"2025-05-28T20:21:31.498283Z","shell.execute_reply.started":"2025-05-28T20:21:22.974299Z","shell.execute_reply":"2025-05-28T20:21:31.497372Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 8. Correlation Visualization","metadata":{}},{"cell_type":"code","source":"# Top correlations bar plot\ntop_20 = corr_df.head(20)\n\nplt.figure(figsize=(12, 8))\ncolors = ['red' if x < 0 else '#0010d9' for x in top_20['Correlation']]\nplt.barh(range(len(top_20)), top_20['Correlation'], color=colors, alpha=0.7)\nplt.yticks(range(len(top_20)), top_20['Feature'])\nplt.xlabel('Correlation with Target')\nplt.title('Top 20 Features - Correlation with Target')\nplt.axvline(x=0, color='black', linestyle='-', alpha=0.3)\nplt.grid(True, alpha=0.3)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T20:21:31.499190Z","iopub.execute_input":"2025-05-28T20:21:31.499474Z","iopub.status.idle":"2025-05-28T20:21:31.747397Z","shell.execute_reply.started":"2025-05-28T20:21:31.499448Z","shell.execute_reply":"2025-05-28T20:21:31.746530Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Key Findings:\n\n* Very weak correlations overall: All correlations are between -0.06 and +0.06, indicating extremely weak linear relationships\n* Slight positive bias: Most features (19 out of 20) show small positive correlations\n* Single negative correlation: Only X531 shows a negative correlation (~-0.04)\n* Uniformly low predictive power: The weak correlations suggest linear models may struggle with this dataset","metadata":{}},{"cell_type":"code","source":"# Correlation distribution\nplt.figure(figsize=(10, 8))\nplt.hist(corr_df['Correlation'], bins=50, alpha=0.7, color='skyblue', edgecolor='black')\nplt.axvline(corr_df['Correlation'].mean(), color='red', linestyle='--', \n           label=f'Mean: {corr_df[\"Correlation\"].mean():.4f}')\nplt.xlabel('Correlation with Target')\nplt.ylabel('Frequency')\nplt.title('Distribution of Feature Correlations')\nplt.legend()\nplt.grid(True, alpha=0.3)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T20:21:31.748193Z","iopub.execute_input":"2025-05-28T20:21:31.748428Z","iopub.status.idle":"2025-05-28T20:21:31.989845Z","shell.execute_reply.started":"2025-05-28T20:21:31.748412Z","shell.execute_reply":"2025-05-28T20:21:31.988738Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n### Distribution Properties:\n\n* Centered near zero: Mean correlation is 0.0091, very close to zero\n* Symmetric distribution: The histogram shows a roughly normal distribution of correlations\n* Narrow range: Most correlations fall within ±0.04, confirming the weak relationships\n* Consistent with the correlations bar plot: This validates the findings from the individual feature correlation chart","metadata":{}},{"cell_type":"markdown","source":"## 9. Feature Distribution Analysis","metadata":{}},{"cell_type":"code","source":"# Analyze top 12 features distributions\ntop_features = corr_df.head(12)['Feature'].tolist()\n\nfig, axes = plt.subplots(3, 4, figsize=(20, 15))\nfig.suptitle('Top 12 Features - Distribution Analysis', fontsize=16)\naxes = axes.flatten()\n\nfor i, feature in enumerate(top_features):\n    data = train[feature].dropna()\n    \n    # Histogram\n    axes[i].hist(data, bins=30, alpha=0.7, color='lightblue', edgecolor='black')\n    \n    # Add mean and median lines\n    mean_val = data.mean()\n    median_val = data.median()\n    axes[i].axvline(mean_val, color='red', linestyle='--', alpha=0.8, label=f'Mean: {mean_val:.3f}')\n    axes[i].axvline(median_val, color='green', linestyle='--', alpha=0.8, label=f'Median: {median_val:.3f}')\n    \n    axes[i].set_title(f'{feature}\\nSkew: {data.skew():.3f}')\n    axes[i].set_xlabel('Value')\n    axes[i].set_ylabel('Frequency')\n    axes[i].legend(fontsize=8)\n    axes[i].grid(True, alpha=0.3)\n\nplt.tight_layout()\nplt.show()","metadata":{"_uuid":"9c16d377-a85c-41f6-af0d-b4e54b661117","_cell_guid":"6bc90026-ca8f-4928-9154-2ed87bbc9722","trusted":true,"execution":{"iopub.status.busy":"2025-05-28T20:21:31.990622Z","iopub.execute_input":"2025-05-28T20:21:31.990851Z","iopub.status.idle":"2025-05-28T20:21:34.604854Z","shell.execute_reply.started":"2025-05-28T20:21:31.990834Z","shell.execute_reply":"2025-05-28T20:21:34.603871Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 10. Feature Statistics Summary\n\n","metadata":{}},{"cell_type":"code","source":"print(\"=== FEATURE STATISTICS SUMMARY ===\")\n\n# Get top 20 features for detailed analysis\ntop_20_features = corr_df.head(20)['Feature'].tolist()\n\n# Create summary statistics\nsummary_stats = []\nfor feature in top_20_features:\n    data = train[feature].dropna()\n    stats_dict = {\n        'Feature': feature,\n        'Count': len(data),\n        'Mean': data.mean(),\n        'Std': data.std(),\n        'Min': data.min(),\n        'Max': data.max(),\n        'Skewness': data.skew(),\n        'Kurtosis': data.kurtosis(),\n        'Correlation': corr_df[corr_df['Feature'] == feature]['Correlation'].iloc[0]\n    }\n    summary_stats.append(stats_dict)\n\nsummary_df = pd.DataFrame(summary_stats)\nprint(\"Top 20 Features Summary Statistics:\")\nsummary_df.round(4)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T20:21:34.605602Z","iopub.execute_input":"2025-05-28T20:21:34.605855Z","iopub.status.idle":"2025-05-28T20:21:34.910326Z","shell.execute_reply.started":"2025-05-28T20:21:34.605835Z","shell.execute_reply":"2025-05-28T20:21:34.909567Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 11. Correlation Heatmap","metadata":{}},{"cell_type":"code","source":"# Create correlation heatmap for top features\ntop_15_features = corr_df.head(15)['Feature'].tolist() + ['label']\n\n# Sample data if too large\nif len(train) > 5000:\n    sample_data = train[top_15_features].sample(n=5000, random_state=42)\nelse:\n    sample_data = train[top_15_features]\n\ncorrelation_matrix = sample_data.corr()\n\nplt.figure(figsize=(12, 10))\nmask = np.triu(np.ones_like(correlation_matrix, dtype=bool))\n\nsns.heatmap(correlation_matrix, \n            mask=mask,\n            annot=True, \n            cmap='crest', \n            center=0,\n            square=True, \n            fmt='.3f',\n            cbar_kws={\"shrink\": .8})\n\nplt.title('Correlation Heatmap - Top 15 Features + Target', fontsize=14)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T20:21:34.911079Z","iopub.execute_input":"2025-05-28T20:21:34.911290Z","iopub.status.idle":"2025-05-28T20:21:35.414293Z","shell.execute_reply.started":"2025-05-28T20:21:34.911272Z","shell.execute_reply":"2025-05-28T20:21:35.413662Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Key observation:\n* **Strong positive** correlations with target: X287 (0.998), X289 (0.995), and X291 (0.987) show extremely high correlations\n* **High inter-feature** correlations: Many features are highly correlated with each other (red squares), suggesting potential multicollinearity\n* **Negative correlations**: Some features like X863, X219, X860, and X531 show weak negative correlations with others\n* **Feature clusters**: You can see blocks of highly correlated features, particularly among X20, X28, X29, X19, X27, X22","metadata":{}},{"cell_type":"markdown","source":"## 12. Feature Value Ranges Analysis\n","metadata":{}},{"cell_type":"code","source":"print(\"=== FEATURE VALUE RANGES ANALYSIS ===\")\n\n# Analyze value ranges for top features\ntop_10_features = corr_df.head(10)['Feature'].tolist()\n\nrange_analysis = []\nfor feature in top_10_features:\n    data = train[feature].dropna()\n    range_dict = {\n        'Feature': feature,\n        'Min': data.min(),\n        'Max': data.max(),\n        'Range': data.max() - data.min(),\n        '1st_Quartile': data.quantile(0.25),\n        '3rd_Quartile': data.quantile(0.75),\n        'IQR': data.quantile(0.75) - data.quantile(0.25),\n        'Correlation': corr_df[corr_df['Feature'] == feature]['Correlation'].iloc[0]\n    }\n    range_analysis.append(range_dict)\n\nrange_df = pd.DataFrame(range_analysis)\nprint(\"Top 10 Features - Value Ranges:\")\nprint(range_df.round(6))\n\n# Visualize ranges\nplt.figure(figsize=(12, 8))\nfeatures = range_df['Feature']\nranges = range_df['Range']\ncolors = ['red' if x < 0 else '#0010d9' for x in range_df['Correlation']]\n\nplt.barh(range(len(features)), ranges, color=colors, alpha=0.7)\nplt.yticks(range(len(features)), features)\nplt.xlabel('Value Range')\nplt.title('Top 10 Features - Value Ranges (Color = Correlation Sign)')\nplt.grid(True, alpha=0.3)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T20:21:35.415258Z","iopub.execute_input":"2025-05-28T20:21:35.415621Z","iopub.status.idle":"2025-05-28T20:21:35.916447Z","shell.execute_reply.started":"2025-05-28T20:21:35.415595Z","shell.execute_reply":"2025-05-28T20:21:35.915741Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Key observations\n* X19 has the largest value range (~27), making it the most variable feature\n* X29 and X28 also have large ranges (~27 and ~25 respectively)\n* X858 has a moderate range (~13)\n* Several features (X219, X22, X863, X20, X21) have smaller, more similar ranges (roughly 6-12)","metadata":{}},{"cell_type":"markdown","source":"## 13. Data Quality Assessment","metadata":{}},{"cell_type":"code","source":" print(\"=== DATA QUALITY ASSESSMENT ===\")\n\n# Check for constant features\nconstant_features = []\nfor col in train.columns:\n    if col != 'label':\n        if train[col].nunique() <= 1:\n            constant_features.append(col)\n\nprint(f\"Constant features (nunique <= 1): {len(constant_features)}\")\nif constant_features:\n    print(\"Constant features:\", constant_features[:10])\n\n# Check for highly skewed features\nhighly_skewed = []\nfor feature in corr_df.head(20)['Feature']:\n    skew_val = train[feature].skew()\n    if abs(skew_val) > 2:\n        highly_skewed.append((feature, skew_val))\n\nprint(f\"\\nHighly skewed features (|skew| > 2): {len(highly_skewed)}\")\nfor feature, skew_val in highly_skewed[:10]:\n    print(f\"  {feature}: {skew_val:.3f}\")\n\n# Check feature correlations between themselves\nprint(f\"\\nInter-feature correlation analysis:\")\ntop_features_for_corr = corr_df.head(10)['Feature'].tolist()\nfeature_corr_matrix = train[top_features_for_corr].corr()\n\n# Find highly correlated feature pairs\nhigh_corr_pairs = []\nfor i in range(len(feature_corr_matrix.columns)):\n    for j in range(i+1, len(feature_corr_matrix.columns)):\n        corr_val = feature_corr_matrix.iloc[i, j]\n        if abs(corr_val) > 0.8:  # High correlation threshold\n            high_corr_pairs.append((\n                feature_corr_matrix.columns[i], \n                feature_corr_matrix.columns[j], \n                corr_val\n            ))\n\nprint(f\"Highly correlated feature pairs (|r| > 0.8): {len(high_corr_pairs)}\")\nfor feat1, feat2, corr_val in high_corr_pairs:\n    print(f\"  {feat1} <-> {feat2}: {corr_val:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T20:21:35.917198Z","iopub.execute_input":"2025-05-28T20:21:35.917389Z","iopub.status.idle":"2025-05-28T20:21:54.257536Z","shell.execute_reply.started":"2025-05-28T20:21:35.917376Z","shell.execute_reply":"2025-05-28T20:21:54.256738Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 14. Final Summary and Insights\n\n","metadata":{}},{"cell_type":"code","source":"print(\"=\" * 60)\nprint(\"🎯 FINAL EDA SUMMARY AND INSIGHTS\")\nprint(\"=\" * 60)\n\nprint(f\"📊 Dataset Overview:\")\nprint(f\"  • Training samples: {len(train):,}\")\nprint(f\"  • Test samples: {len(test):,}\")\nprint(f\"  • Total features: {len([col for col in train.columns if col != 'label'])}\")\nprint(f\"  • Target variable: 'label'\")\n\nprint(f\"\\n🎯 Target Variable:\")\nprint(f\"  • Mean: {target.mean():.6f}\")\nprint(f\"  • Std: {target.std():.6f}\")\nprint(f\"  • Skewness: {target.skew():.3f}\")\nprint(f\"  • Distribution: {'Normal' if abs(target.skew()) < 0.5 else 'Skewed'}\")\n\nprint(f\"\\n🔍 Data Quality:\")\nmissing_cols = len([col for col in train.columns if train[col].isnull().sum() > 0])\nprint(f\"  • Columns with missing values: {missing_cols}\")\nprint(f\"  • Constant features: {len(constant_features)}\")\nprint(f\"  • Highly skewed features: {len(highly_skewed)}\")\n\nprint(f\"\\n📈 Feature Correlations:\")\nprint(f\"  • Features with strong correlation (|r| > 0.1): {len(strong_corr)}\")\nprint(f\"  • Features with moderate correlation (0.05 < |r| ≤ 0.1): {len(moderate_corr)}\")\nprint(f\"  • Features with weak correlation (|r| ≤ 0.05): {len(weak_corr)}\")\n\nprint(f\"\\n🏆 Top 5 Most Important Features:\")\nfor i, row in corr_df.head(5).iterrows():\n    print(f\"  {i+1}. {row['Feature']}: {row['Correlation']:.6f}\")\n\nprint(f\"\\n💡 Key Insights:\")\nprint(f\"  • Target shows {'low' if target.std() < 0.01 else 'moderate' if target.std() < 0.1 else 'high'} variability\")\nprint(f\"  • {'Few' if len(strong_corr) < 10 else 'Many'} features show strong correlation with target\")\nprint(f\"  • Data appears {'clean' if missing_cols == 0 else 'to have missing values'}\")\nprint(f\"  • Feature engineering {'may' if len(highly_skewed) > 5 else 'might not'} be needed for skewed features\")\n\nprint(\"\\n✅ EDA Complete! feel free to create a copy and share your insights in the comments !\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T20:21:54.258446Z","iopub.execute_input":"2025-05-28T20:21:54.258660Z","iopub.status.idle":"2025-05-28T20:21:54.955090Z","shell.execute_reply.started":"2025-05-28T20:21:54.258642Z","shell.execute_reply":"2025-05-28T20:21:54.954251Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}