{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd \nimport numpy as np","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-22T17:30:35.620955Z","iopub.execute_input":"2025-07-22T17:30:35.621438Z","iopub.status.idle":"2025-07-22T17:30:38.271142Z","shell.execute_reply.started":"2025-07-22T17:30:35.621404Z","shell.execute_reply":"2025-07-22T17:30:38.269278Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Cryptocurrency Market Prediction Project\n# 📋 Project Overview\nThis project predicts cryptocurrency market movements using machine learning. We use historical trading data to forecast future price trends.\n\n🛠️ What This Code Does\n# 1. Data Loading\nReads training and test data from Parquet files\n\nLoads sample submission format\n\nUses data from Kaggle competition environment\n\n# 2. Feature Selection\nWe start with important market features:\n\nBasic trading data: bid/ask quantities, buy/sell volumes, total volume\n\nSelected X features: 40+ market indicators (X363, X321, X405, etc.)\n\n# 3. Feature Engineering\nWe create 50+ advanced market indicators:\n\nPrice Pressure Indicators:\n\nnet_order_flow: Difference between buying and selling\n\nbuying_pressure: How much buying is happening relative to total volume\n\nLiquidity Measures:\n\ntotal_depth: Total available market liquidity\n\ndepth_imbalance: Difference between buy and sell liquidity\n\nMarket Activity:\n\nactivity_intensity: How actively the market is trading\n\nkyle_lambda: Measures market impact of trades\n\nRisk Indicators:\n\nrealized_spread_proxy: Trading cost estimation\n\nexecution_shortfall_proxy: Price movement during trade execution\n\n# 4. Final Feature Selection\nWe choose 25 best-performing features including:\n\n15 engineered features (normalized_net_flow, buying_pressure, etc.)\n\n10 original X features (X779, X752, X759, etc.)\n\n# 5. Data Preprocessing\nSorting: Arrange data in time order\n\nScaling: Normalize all features for better model performance\n\nSplitting: 90% for training, 10% for validation\n\n# 6. Machine Learning Model\nWe use MLPRegressor (Neural Network) with:\n\nArchitecture: 2 hidden layers (1200 → 100 neurons)\n\nTraining: 6 iterations with early stopping\n\nOptimization: Adam optimizer with adaptive learning rate\n\n# 📊 Results\nTraining Performance:\nPearson Correlation: 0.1421 (positive relationship)\n\nMean Squared Error: 0.9903\n\nValidation Performance:\nPearson Correlation: 0.1766 (better than training)\n\nMean Squared Error: 1.0791\n\n# 🎯 Key Insights\nValidation performs better than training (0.1766 vs 0.1421 correlation)\n\nModel is learning patterns without overfitting\n\nFeature engineering helped create meaningful market indicators\n\nThe model shows positive predictive power despite market complexity\n\n","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport warnings\nfrom sklearn.metrics import mean_squared_error\nfrom scipy.stats import pearsonr\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.neural_network import MLPRegressor # Import MLPRegressor\n\nwarnings.filterwarnings('ignore') # Suppress warnings for cleaner output\n\n# --- Load the datasets ---\ntry:\n    train_df = pd.read_parquet(\"train.parquet\")\n    test_df = pd.read_parquet(\"test.parquet\")\n    submission_df = pd.read_csv(\"sample_submission.csv\")\nexcept FileNotFoundError:\n    print(\"Ensure train.parquet, test.parquet, and sample_submission.csv are in the correct path (e.g., /kaggle/input/drw-crypto-market-prediction/)\")\n    # Fallback for Kaggle environment\n    train_df = pd.read_parquet(\"/kaggle/input/drw-crypto-market-prediction/train.parquet\")\n    test_df = pd.read_parquet(\"/kaggle/input/drw-crypto-market-prediction/test.parquet\")\n    submission_df = pd.read_csv(\"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\")\n\n# Define base features required to derive ANY of the engineered features.\nbase_features_for_derivation = [\n    \"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\", \"label\"]\n\n# List of all \"X\" features to be included as per new instructions (original set, not the filtered one yet)\nx_features_to_include_new = [\n    'X363', 'X321', 'X405', 'X730', 'X523', 'X756', 'X589', 'X462', 'X779',\n    'X25', 'X532', 'X520', 'X329', 'X383',\n    \"X752\", \"X287\", \"X298\", \"X759\", \"X302\", \"X55\", \"X56\", \"X52\", \"X303\",\n    \"X51\", \"X598\", \"X385\", \"X603\", \"X674\", \"X415\", \"X345\", \"X174\", \"X178\",\n    \"X168\", \"X612\", \"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"\n]\n\n# Combine X features with base features for initial loading.\n# Ensure 'label' is only in the training set loading if it's not a feature for prediction.\n# For test_df, 'label' should not be loaded.\nall_features_to_load_initial_train = list(set(x_features_to_include_new + [\"label\"]))\nall_features_to_load_initial_test = list(set(x_features_to_include_new)) # No 'label' for test\n\n# Select only the necessary features for train and test sets.\ntrain_df_processed = train_df[all_features_to_load_initial_train].copy()\ntest_df_processed = test_df[all_features_to_load_initial_test].copy()\n\n\n# -----------------------------------------------------------------------------\n## Feature Engineering (Create all potential features)\n# -----------------------------------------------------------------------------\ndef feature_engineering(df):\n    # Ensure raw quantities exist before creating interactions\n    for col in [\"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"]:\n        if col not in df.columns:\n            df[col] = 0.0 # Assign 0.0 to handle missing columns if any\n\n    # Original features\n    df['bid_ask_interaction'] = df['bid_qty'] * df['ask_qty']\n    df['bid_buy_interaction'] = df['bid_qty'] * df['buy_qty']\n    df['bid_sell_interaction'] = df['bid_qty'] * df['sell_qty']\n    df['ask_buy_interaction'] = df['ask_qty'] * df['buy_qty']\n    df['ask_sell_interaction'] = df['ask_qty'] * df['sell_qty']\n\n    df['volume_weighted_sell'] = df['sell_qty'] * df['volume']\n    df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-10)\n    df['selling_pressure'] = df['sell_qty'] / (df['volume'] + 1e-10)\n    df['log_volume'] = np.log1p(df['volume'])\n\n    df['effective_spread_proxy'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-10)\n    df['bid_ask_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['liquidity_ratio'] = (df['bid_qty'] + df['ask_qty']) / (df['volume'] + 1e-10)\n    \n    # === NEW MICROSTRUCTURE FEATURES ===\n    \n    # Price Pressure Indicators\n    df['net_order_flow'] = df['buy_qty'] - df['sell_qty']\n    df['normalized_net_flow'] = df['net_order_flow'] / (df['volume'] + 1e-10)\n    df['buying_pressure'] = df['buy_qty'] / (df['volume'] + 1e-10)\n    df['volume_weighted_buy'] = df['buy_qty'] * df['volume']\n    \n    # Liquidity Depth Measures\n    df['total_depth'] = df['bid_qty'] + df['ask_qty']\n    df['depth_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['total_depth'] + 1e-10)\n    df['relative_spread'] = np.abs(df['bid_qty'] - df['ask_qty']) / (df['total_depth'] + 1e-10)\n    df['log_depth'] = np.log1p(df['total_depth'])\n    \n    # Order Flow Toxicity Proxies\n    df['kyle_lambda'] = np.abs(df['net_order_flow']) / (df['volume'] + 1e-10)\n    df['flow_toxicity'] = np.abs(df['order_flow_imbalance']) * df['volume']\n    df['aggressive_flow_ratio'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    \n    # Market Activity Indicators\n    df['volume_depth_ratio'] = df['volume'] / (df['total_depth'] + 1e-10)\n    df['activity_intensity'] = (df['buy_qty'] + df['sell_qty']) / (df['volume'] + 1e-10)\n    df['log_buy_qty'] = np.log1p(df['buy_qty'])\n    df['log_sell_qty'] = np.log1p(df['sell_qty'])\n    df['log_bid_qty'] = np.log1p(df['bid_qty'])\n    df['log_ask_qty'] = np.log1p(df['ask_qty'])\n    \n    # Microstructure Volatility Proxies\n    df['realized_spread_proxy'] = 2 * np.abs(df['net_order_flow']) / (df['volume'] + 1e-10)\n    df['price_impact_proxy'] = df['net_order_flow'] / (df['total_depth'] + 1e-10)\n    df['quote_volatility_proxy'] = np.abs(df['depth_imbalance'])\n    \n    # Complex Interaction Terms\n    df['flow_depth_interaction'] = df['net_order_flow'] * df['total_depth']\n    df['imbalance_volume_interaction'] = df['order_flow_imbalance'] * df['volume']\n    df['depth_volume_interaction'] = df['total_depth'] * df['volume']\n    df['buy_sell_spread'] = np.abs(df['buy_qty'] - df['sell_qty'])\n    df['bid_ask_spread'] = np.abs(df['bid_qty'] - df['ask_qty'])\n    \n    # Information Asymmetry Measures\n    df['trade_informativeness'] = df['net_order_flow'] / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['execution_shortfall_proxy'] = df['buy_sell_spread'] / (df['volume'] + 1e-10)\n    df['adverse_selection_proxy'] = df['net_order_flow'] / (df['total_depth'] + 1e-10) * df['volume']\n    \n    # Market Efficiency Indicators\n    df['fill_probability'] = df['volume'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['execution_rate'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    df['market_efficiency'] = df['volume'] / (df['bid_ask_spread'] + 1e-10)\n    \n    # Non-linear Transformations\n    df['sqrt_volume'] = np.sqrt(df['volume'])\n    df['sqrt_depth'] = np.sqrt(df['total_depth'])\n    df['volume_squared'] = df['volume'] ** 2\n    df['imbalance_squared'] = df['order_flow_imbalance'] ** 2\n    \n    # Relative Measures\n    df['bid_ratio'] = df['bid_qty'] / (df['total_depth'] + 1e-10)\n    df['ask_ratio'] = df['ask_qty'] / (df['total_depth'] + 1e-10)\n    df['buy_ratio'] = df['buy_qty'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['sell_ratio'] = df['sell_qty'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    \n    # Market Stress Indicators\n    df['liquidity_consumption'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    df['market_stress'] = df['volume'] / (df['total_depth'] + 1e-10) * np.abs(df['order_flow_imbalance'])\n    df['depth_depletion'] = df['volume'] / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    \n    # Directional Indicators\n    df['net_buying_ratio'] = df['net_order_flow'] / (df['volume'] + 1e-10)\n    df['directional_volume'] = df['net_order_flow'] * np.log1p(df['volume'])\n    df['signed_volume'] = np.sign(df['net_order_flow']) * df['volume']\n    \n    # Handle infinities and NaN\n    df = df.replace([np.inf, -np.inf], np.nan)\n    \n    # For each column, replace NaN with median for robustness\n    for col in df.columns:\n        if df[col].isna().any():\n            median_val = df[col].median()\n            df[col] = df[col].fillna(median_val if not pd.isna(median_val) else 0)\n    \n    return df\n\nprint(\"--- Starting Feature Engineering (creating all potential features) ---\")\nx_train_full = feature_engineering(train_df_processed.copy())\nx_test_full = feature_engineering(test_df_processed.copy())\nprint(\"--- Feature Engineering Complete ---\")\n\n# Define the NEW set of features based on your selection criteria\n# These are the features with similar statistical properties AND high Pearson correlation\nselected_features_for_model = [\n    'X779', 'normalized_net_flow', 'buying_pressure', 'depth_imbalance',\n    'kyle_lambda', 'activity_intensity', 'realized_spread_proxy',\n    'execution_shortfall_proxy', 'fill_probability', 'imbalance_squared',\n    'bid_ratio', 'ask_ratio', 'buy_ratio', 'sell_ratio', 'net_buying_ratio',\n    'X752', 'X759', 'X756', 'X287', 'X298', 'X302', 'X303', 'X385', 'X52'\n]\n\n# Use ONLY these selected features for training and testing\nfinal_features_to_use = selected_features_for_model\n\n# Filter x_train_full and x_test_full to contain ONLY these chosen features\nX_full_train = x_train_full[final_features_to_use].copy()\ny_full_train = x_train_full['label'].copy()\nx_test_full = x_test_full[final_features_to_use].copy()\n\n# Sort the training dataset by its implicit timestamp index (assuming index preserves time)\n# This is crucial for chronological split.\nX_full_train = X_full_train.sort_index()\ny_full_train = y_full_train.sort_index()\n\n# -----------------------------------------------------------------------------\n## Data Preprocessing (Scaling and Splitting)\n# -----------------------------------------------------------------------------\n# Identify numerical columns for scaling.\nnumerical_cols_to_process = X_full_train.columns.tolist()\n\n# Initialize StandardScaler\nscaler = StandardScaler()\nprint(\"--- Starting Data Normalization ---\")\nX_full_train[numerical_cols_to_process] = scaler.fit_transform(X_full_train[numerical_cols_to_process])\nx_test_full[numerical_cols_to_process] = scaler.transform(x_test_full[numerical_cols_to_process])\nprint(\"--- Data Normalization Complete ---\")\n\n# Define the chronological split point for training and validation data (e.g., 90% train, 10% validation)\nsplit_index = int(len(X_full_train) * 0.9)\n\n# Create chronological train and validation sets\nX_train_split = X_full_train.iloc[:split_index].values\ny_train_split = y_full_train.iloc[:split_index].values\nX_val_split = X_full_train.iloc[split_index:].values\ny_val_split = y_full_train.iloc[split_index:].values\n\n# Test set for final prediction (already scaled and feature-engineered)\nx_test_np = x_test_full.values\n\n# -----------------------------------------------------------------------------\n## MLPRegressor Model Definition and Training (Fit Only Once)\n# -----------------------------------------------------------------------------\nprint(\"\\n--- Building and Training MLPRegressor Model (Fit Once) ---\")\n# Initialize the MLPRegressor model\n# Hyperparameters for MLPRegressor. These are common starting points.\n# You will likely need to tune these for optimal performance.\nmlp_model = MLPRegressor(hidden_layer_sizes=(1200, 100), # Two hidden layers with 100 and 50 neurons\n                         activation='identity',          # ReLU activation function\n                         solver='adam',                  # Adam optimizer\n                         alpha=1,                        # L2 regularization parameter (strength of regularization)\n                         batch_size='auto',              # Use automatic batch size\n                         learning_rate='adaptive',       # Adjusts learning rate based on validation loss\n                         max_iter=6,                    # Maximum number of iterations (epochs)\n                         early_stopping=True,            # Use early stopping based on validation score\n                         n_iter_no_change=20,            # Number of epochs with no improvement to wait before early stopping\n                         validation_fraction=0.01,       # Fraction of training data to set aside for validation\n                         random_state=42,                # For reproducibility\n                         verbose=True                    # Print progress messages to stdout\n                        )\n\n# Train the model on the chronological training split.\n# MLPRegressor's `early_stopping` uses an internal validation split.\n# If you want to use your pre-defined X_val_split, you would need a custom callback or manual loop,\n# but for simplicity, we'll let `MLPRegressor` handle its own internal split for early stopping.\nmlp_model.fit(X_train_split, y_train_split)\nprint(\"--- MLPRegressor Model Training Complete (Fit Once) ---\")\n\n# -----------------------------------------------------------------------------\n## Evaluation on Training and Validation Sets (from the single fit)\n# -----------------------------------------------------------------------------\n# Make predictions on training and validation sets using the fitted model\ny_train_pred_mlp = mlp_model.predict(X_train_split)\ny_val_pred_mlp = mlp_model.predict(X_val_split)\n\nprint(\"\\n📊 **Training Set Metrics (MLPRegressor):**\")\npearson_corr_train_mlp, _ = pearsonr(y_train_split, y_train_pred_mlp)\nprint(f\"✅ Pearson Correlation: {pearson_corr_train_mlp:.4f}\")\nmse_train_mlp = mean_squared_error(y_train_split, y_train_pred_mlp)\nprint(f\"📉 Mean Squared Error (MSE): {mse_train_mlp:.4f}\")\n\nprint(\"\\n📊 **Validation Set Metrics (MLPRegressor):**\")\npearson_corr_val_mlp, _ = pearsonr(y_val_split, y_val_pred_mlp)\nprint(f\"✅ Pearson Correlation: {pearson_corr_val_mlp:.4f}\")\nmse_val_mlp = mean_squared_error(y_val_split, y_val_pred_mlp)\nprint(f\"📉 Mean Squared Error (MSE): {mse_val_mlp:.4f}\")\n\n# -----------------------------------------------------------------------------\n## Final Prediction on Test Set (using the same model, fitted only once)\n# -----------------------------------------------------------------------------\nprint(\"\\n--- Generating predictions on test data using the single fitted MLPRegressor model ---\")\ny_pred_mlp = mlp_model.predict(x_test_np)\n\n# -----------------------------------------------------------------------------\n## Create Submission File\n# -----------------------------------------------------------------------------\nsubmission_df[\"prediction\"] = y_pred_mlp\nsubmission_df.to_csv(\"submission.csv\", index=False)\nprint(\"\\n📁 Submission file saved as 'submission.csv'\")\nprint(submission_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-22T17:30:38.273913Z","iopub.execute_input":"2025-07-22T17:30:38.274663Z","iopub.status.idle":"2025-07-22T17:33:50.980666Z","shell.execute_reply.started":"2025-07-22T17:30:38.274611Z","shell.execute_reply":"2025-07-22T17:33:50.979529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}