{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-21T08:39:47.200652Z","iopub.execute_input":"2025-07-21T08:39:47.201003Z","iopub.status.idle":"2025-07-21T08:39:47.207731Z","shell.execute_reply.started":"2025-07-21T08:39:47.200952Z","shell.execute_reply":"2025-07-21T08:39:47.206673Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.preprocessing import StandardScaler, RobustScaler\nfrom sklearn.model_selection import TimeSeriesSplit, cross_val_score\nfrom sklearn.metrics import mean_squared_error, mean_absolute_error\nfrom sklearn.feature_selection import SelectKBest, f_regression, mutual_info_regression\nfrom sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.linear_model import Ridge, Lasso, ElasticNet\nimport lightgbm as lgb\nimport xgboost as xgb\nfrom scipy import stats\nfrom scipy.stats import pearsonr\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T08:39:47.209087Z","iopub.execute_input":"2025-07-21T08:39:47.209373Z","iopub.status.idle":"2025-07-21T08:39:57.711904Z","shell.execute_reply.started":"2025-07-21T08:39:47.209347Z","shell.execute_reply":"2025-07-21T08:39:57.710886Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"try:\n    import torch\n    import torch.nn as nn\n    import torch.optim as optim\n    from torch.utils.data import DataLoader, TensorDataset\n    PYTORCH_AVAILABLE = True\nexcept ImportError:\n    PYTORCH_AVAILABLE = False\n    print(\"PyTorch not available. Using tree-based models only.\")\n\n# Set random seeds for reproducibility\nnp.random.seed(42)\nif PYTORCH_AVAILABLE:\n    torch.manual_seed(42)\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(\"=== DRW Crypto Price Prediction Challenge ===\")\nprint(\"Comprehensive ML Solution with Ensemble Methods\")\nprint(\"=\"*50)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T08:39:57.712772Z","iopub.execute_input":"2025-07-21T08:39:57.713367Z","iopub.status.idle":"2025-07-21T08:40:03.867353Z","shell.execute_reply.started":"2025-07-21T08:39:57.713341Z","shell.execute_reply":"2025-07-21T08:40:03.866365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\ndef clear_memory():\n        gc.collect()\n        if torch.cuda.is_available():\n            torch.cuda.empty_cache()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T08:40:03.868238Z","iopub.execute_input":"2025-07-21T08:40:03.868899Z","iopub.status.idle":"2025-07-21T08:40:03.873751Z","shell.execute_reply.started":"2025-07-21T08:40:03.868865Z","shell.execute_reply":"2025-07-21T08:40:03.872950Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ============================================================================\n# 1. DATA LOADING AND INITIAL EXPLORATION\n# ============================================================================","metadata":{}},{"cell_type":"code","source":"def load_and_explore_data():\n    \"\"\"Load data and perform initial exploration\"\"\"\n    print(\"\\n1. Loading and Exploring Data...\")\n    \n    # Load training data\n    TRAINING_PATH = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    TESTING_PATH = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    train_df = pd.read_parquet(TRAINING_PATH)\n    test_df = pd.read_parquet(TESTING_PATH)\n\n    start_date = \"2023-11-01 00:00:00\"\n    end_date = \"2024-02-29 23:59:00\"\n\n    # Filter the DataFrame and update train_df with the subset\n    train_df = train_df.loc[start_date:end_date]\n    \n    print(f\"Training data shape: {train_df.shape}\")\n    print(f\"Test data shape: {test_df.shape}\")\n    print(f\"Training period: {train_df.index.min()} to {train_df.index.max()}\")\n\n  # Basic statistics\n    print(\"\\nBasic Statistics:\")\n    print(f\"Target variable (label) statistics:\")\n    print(train_df['label'].describe())\n\n# Check for missing values\n    print(f\"\\nMissing values in training data: {train_df.isnull().sum().sum()}\")\n    print(f\"Missing values in test data: {test_df.isnull().sum().sum()}\")\n\n# Feature columns\n    feature_cols = [col for col in train_df.columns if col.startswith('X')]\n    public_cols = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']\n    \n    print(f\"\\nNumber of proprietary features (X_*): {len(feature_cols)}\")\n    print(f\"Number of public market features: {len(public_cols)}\")\n    \n    return train_df, test_df, feature_cols, public_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T08:40:03.875764Z","iopub.execute_input":"2025-07-21T08:40:03.876037Z","iopub.status.idle":"2025-07-21T08:40:03.896256Z","shell.execute_reply.started":"2025-07-21T08:40:03.876016Z","shell.execute_reply":"2025-07-21T08:40:03.895216Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ============================================================================\n# 2. ADVANCED FEATURE ENGINEERING\n# ============================================================================","metadata":{}},{"cell_type":"code","source":"  def create_market_microstructure_features(df):\n    \"\"\"Create advanced market microstructure features\"\"\"\n    print(\"\\n2. Creating Market Microstructure Features...\")\n    \n    df = df.copy()\n    \n    # Order book imbalance features\n    df['bid_ask_spread'] = df['ask_qty'] - df['bid_qty']\n    df['bid_ask_ratio'] = df['bid_qty'] / (df['ask_qty'] + 1e-8)\n    df['total_liquidity'] = df['bid_qty'] + df['ask_qty']\n    \n    # Trading intensity features\n    df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-8)\n    df['buy_pressure'] = df['buy_qty'] / (df['volume'] + 1e-8)\n    df['sell_pressure'] = df['sell_qty'] / (df['volume'] + 1e-8)\n    df['net_flow'] = df['buy_qty'] - df['sell_qty']\n    \n    # Volume-weighted features\n    df['volume_ma_5'] = df['volume'].rolling(window=5).mean()\n    df['volume_ma_15'] = df['volume'].rolling(window=15).mean()\n    df['volume_ratio'] = df['volume'] / (df['volume_ma_15'] + 1e-8)\n    \n    # Momentum and volatility proxies\n    for col in ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']:\n        df[f'{col}_pct_change'] = df[col].pct_change()\n        df[f'{col}_rolling_std'] = df[col].rolling(window=10).std()\n        df[f'{col}_z_score'] = (df[col] - df[col].rolling(window=20).mean()) / (df[col].rolling(window=20).std() + 1e-8)\n    \n    return df\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T08:40:03.897404Z","iopub.execute_input":"2025-07-21T08:40:03.897729Z","iopub.status.idle":"2025-07-21T08:40:03.921251Z","shell.execute_reply.started":"2025-07-21T08:40:03.897698Z","shell.execute_reply":"2025-07-21T08:40:03.920455Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_technical_indicators(df):\n    \"\"\"Create technical analysis indicators\"\"\"\n    print(\"Creating Technical Indicators...\")\n    \n    df = df.copy()\n    \n    # Price proxy using volume-weighted average\n    df['price_proxy'] = (df['bid_qty'] + df['ask_qty']) / 2\n    \n    # Moving averages\n    for window in [5, 10, 15, 30]:\n        df[f'price_ma_{window}'] = df['price_proxy'].rolling(window=window).mean()\n        df[f'volume_ma_{window}'] = df['volume'].rolling(window=window).mean()\n    \n    # RSI-like indicator\n    delta = df['price_proxy'].diff()\n    gain = (delta.where(delta > 0, 0)).rolling(window=14).mean()\n    loss = (-delta.where(delta < 0, 0)).rolling(window=14).mean()\n    rs = gain / (loss + 1e-8)\n    df['rsi'] = 100 - (100 / (1 + rs))\n    \n    # MACD-like indicator\n    ema_12 = df['price_proxy'].ewm(span=12).mean()\n    ema_26 = df['price_proxy'].ewm(span=26).mean()\n    df['macd'] = ema_12 - ema_26\n    df['macd_signal'] = df['macd'].ewm(span=9).mean()\n    df['macd_histogram'] = df['macd'] - df['macd_signal']\n    \n    # Bollinger Bands\n    df['bb_middle'] = df['price_proxy'].rolling(window=20).mean()\n    bb_std = df['price_proxy'].rolling(window=20).std()\n    df['bb_upper'] = df['bb_middle'] + (bb_std * 2)\n    df['bb_lower'] = df['bb_middle'] - (bb_std * 2)\n    df['bb_position'] = (df['price_proxy'] - df['bb_lower']) / (df['bb_upper'] - df['bb_lower'])\n    \n    return df\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T08:40:03.922961Z","iopub.execute_input":"2025-07-21T08:40:03.923348Z","iopub.status.idle":"2025-07-21T08:40:03.949286Z","shell.execute_reply.started":"2025-07-21T08:40:03.923320Z","shell.execute_reply":"2025-07-21T08:40:03.948091Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_time_features(df):\n    \"\"\"Create time-based features\"\"\"\n    print(\"Creating Time-based Features...\")\n    \n    df = df.copy()\n    df['timestamp'] = df.index\n    if 'timestamp' in df.columns:\n        df['timestamp'] = pd.to_datetime(df['timestamp'])\n        df['hour'] = df['timestamp'].dt.hour\n        df['day_of_week'] = df['timestamp'].dt.dayofweek\n        df['minute'] = df['timestamp'].dt.minute\n        df['is_weekend'] = (df['day_of_week'] >= 5).astype(int)\n        df['is_market_hours'] = ((df['hour'] >= 9) & (df['hour'] <= 16)).astype(int)\n        \n        # Cyclical encoding\n        df['hour_sin'] = np.sin(2 * np.pi * df['hour'] / 24)\n        df['hour_cos'] = np.cos(2 * np.pi * df['hour'] / 24)\n        df['day_sin'] = np.sin(2 * np.pi * df['day_of_week'] / 7)\n        df['day_cos'] = np.cos(2 * np.pi * df['day_of_week'] / 7)\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T08:40:03.950333Z","iopub.execute_input":"2025-07-21T08:40:03.950728Z","iopub.status.idle":"2025-07-21T08:40:03.973664Z","shell.execute_reply.started":"2025-07-21T08:40:03.950699Z","shell.execute_reply":"2025-07-21T08:40:03.972687Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_lag_features(df, feature_cols, lags=[1, 2, 3, 5, 10]):\n    \"\"\"Create lagged features for key variables\"\"\"\n    print(\"Creating Lag Features...\")\n    \n    df = df.copy()\n    \n    # Key features to lag\n    key_features = ['volume', 'bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'net_flow', 'buy_sell_ratio']\n    \n    for feature in key_features:\n        if feature in df.columns:\n            for lag in lags:\n                df[f'{feature}_lag_{lag}'] = df[feature].shift(lag)\n    \n    # Lag some proprietary features (sample a few to avoid overfitting)\n    sample_x_features = feature_cols[:20]  # Use first 20 X features\n    for feature in sample_x_features:\n        for lag in [1, 2, 3]:\n            df[f'{feature}_lag_{lag}'] = df[feature].shift(lag)\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T08:40:03.974700Z","iopub.execute_input":"2025-07-21T08:40:03.975045Z","iopub.status.idle":"2025-07-21T08:40:03.998286Z","shell.execute_reply.started":"2025-07-21T08:40:03.975003Z","shell.execute_reply":"2025-07-21T08:40:03.997314Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_interaction_features(df):\n    \"\"\"Create interaction features between key variables\"\"\"\n    print(\"Creating Interaction Features...\")\n    \n    df = df.copy()\n    \n    # Volume interactions\n    df['volume_bid_interaction'] = df['volume'] * df['bid_qty']\n    df['volume_ask_interaction'] = df['volume'] * df['ask_qty']\n    df['volume_net_flow_interaction'] = df['volume'] * df['net_flow']\n    \n    # Ratio interactions\n    df['bid_ask_volume_ratio'] = df['bid_ask_ratio'] * df['volume_ratio']\n    df['buy_sell_volume_ratio'] = df['buy_sell_ratio'] * df['volume_ratio']\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T08:40:03.999233Z","iopub.execute_input":"2025-07-21T08:40:03.999534Z","iopub.status.idle":"2025-07-21T08:40:04.018239Z","shell.execute_reply.started":"2025-07-21T08:40:03.999506Z","shell.execute_reply":"2025-07-21T08:40:04.017201Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ============================================================================\n# 3. FEATURE SELECTION AND PREPROCESSING\n# ============================================================================","metadata":{}},{"cell_type":"code","source":"def perform_feature_selection(X, y, method='mutual_info', k=200):\n    \"\"\"Perform feature selection using various methods\"\"\"\n    print(f\"\\n3. Performing Feature Selection (method: {method}, k: {k})...\")\n    \n    if method == 'mutual_info':\n        selector = SelectKBest(score_func=mutual_info_regression, k=k)\n    elif method == 'f_regression':\n        selector = SelectKBest(score_func=f_regression, k=k)\n    else:\n        raise ValueError(\"Method must be 'mutual_info' or 'f_regression'\")\n    \n    X_selected = selector.fit_transform(X, y)\n    selected_features = X.columns[selector.get_support()]\n    \n    print(f\"Selected {len(selected_features)} features out of {X.shape[1]}\")\n    \n    return X_selected, selected_features, selector","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T08:40:04.019169Z","iopub.execute_input":"2025-07-21T08:40:04.019498Z","iopub.status.idle":"2025-07-21T08:40:04.042428Z","shell.execute_reply.started":"2025-07-21T08:40:04.019475Z","shell.execute_reply":"2025-07-21T08:40:04.041409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess_data(train_df, test_df , feature_cols, public_cols):\n    \"\"\"Comprehensive data preprocessing pipeline\"\"\"\n    print(\"\\n4. Training Data Preprocessing Pipeline...\")\n    \n    # Feature engineering\n    train_df = create_market_microstructure_features(train_df)\n    train_df = create_technical_indicators(train_df)\n    train_df = create_time_features(train_df)\n    train_df = create_lag_features(train_df, feature_cols)\n    train_df = create_interaction_features(train_df)\n    \n    test_df = create_market_microstructure_features(test_df)\n    test_df = create_technical_indicators(test_df)\n    test_df = create_time_features(test_df)\n    test_df = create_lag_features(test_df, feature_cols)\n    test_df = create_interaction_features(test_df)\n    \n    # Get all feature columns\n    all_feature_cols = [col for col in train_df.columns if col not in ['timestamp', 'label']]\n    \n    # Handle missing values\n    train_df[all_feature_cols] = train_df[all_feature_cols].fillna(method='ffill').fillna(0)\n    test_df[all_feature_cols] = test_df[all_feature_cols].fillna(method='ffill').fillna(0)\n    \n    # Remove infinite values\n    train_df = train_df.replace([np.inf, -np.inf], np.nan).fillna(0)\n    test_df = test_df.replace([np.inf, -np.inf], np.nan).fillna(0)\n    \n    print(f\"Total features after engineering: {len(all_feature_cols)}\")\n\n    try:\n        train_df.to_csv('/kaggle/working/processed_train_data.csv')\n        test_df.to_csv('/kaggle/working/processed_test_data.csv')\n        print(\"Processed data saved successfully to 'processed_train_data.csv'\")\n    except Exception as e:\n        print(f\"Error saving data to CSV: {e}\")\n        \n    return train_df, test_df , all_feature_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T08:40:04.043406Z","iopub.execute_input":"2025-07-21T08:40:04.043774Z","iopub.status.idle":"2025-07-21T08:40:04.063784Z","shell.execute_reply.started":"2025-07-21T08:40:04.043742Z","shell.execute_reply":"2025-07-21T08:40:04.062849Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ============================================================================\n# 4. MODEL DEFINITIONS\n# ============================================================================","metadata":{}},{"cell_type":"code","source":"class LSTMModel(nn.Module):\n    \"\"\"LSTM model for sequence prediction\"\"\"\n    def __init__(self, input_size, hidden_size=64, num_layers=2, dropout=0.3):\n        super(LSTMModel, self).__init__()\n        self.hidden_size = hidden_size\n        self.num_layers = num_layers\n        \n        self.lstm = nn.LSTM(input_size, hidden_size, num_layers, \n                           batch_first=True, dropout=dropout)\n        self.fc = nn.Linear(hidden_size, 1)\n        self.dropout = nn.Dropout(dropout)\n        \n    def forward(self, x):\n        h0 = torch.zeros(self.num_layers, x.size(0), self.hidden_size).to(device)\n        c0 = torch.zeros(self.num_layers, x.size(0), self.hidden_size).to(device)\n        \n        out, _ = self.lstm(x, (h0, c0))\n        out = self.dropout(out[:, -1, :])\n        out = self.fc(out)\n        return out","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T08:40:04.064812Z","iopub.execute_input":"2025-07-21T08:40:04.065162Z","iopub.status.idle":"2025-07-21T08:40:04.089100Z","shell.execute_reply.started":"2025-07-21T08:40:04.065131Z","shell.execute_reply":"2025-07-21T08:40:04.088097Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_sequences(data, seq_length=30):\n    \"\"\"Create sequences for LSTM training\"\"\"\n    sequences = []\n    targets = []\n    \n    for i in range(len(data) - seq_length):\n        seq = data[i:i+seq_length]\n        target = data[i+seq_length]\n        sequences.append(seq)\n        targets.append(target)\n    \n    return np.array(sequences), np.array(targets)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T08:40:04.091436Z","iopub.execute_input":"2025-07-21T08:40:04.091859Z","iopub.status.idle":"2025-07-21T08:40:04.115346Z","shell.execute_reply.started":"2025-07-21T08:40:04.091827Z","shell.execute_reply":"2025-07-21T08:40:04.114461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_lstm_model(X_train, y_train, X_val, y_val, seq_length=30, epochs=50):\n    \"\"\"Train LSTM model\"\"\"\n    print(\"Training LSTM Model...\")\n    \n    if not PYTORCH_AVAILABLE:\n        print(\"PyTorch not available. Skipping LSTM training.\")\n        return None, None\n    \n    # Create sequences\n    X_train_seq, y_train_seq = create_sequences(X_train, seq_length)\n    X_val_seq, y_val_seq = create_sequences(X_val, seq_length)\n    \n    # Convert to tensors\n    X_train_tensor = torch.FloatTensor(X_train_seq).to(device)\n    y_train_tensor = torch.FloatTensor(y_train_seq).unsqueeze(1).to(device)\n    X_val_tensor = torch.FloatTensor(X_val_seq).to(device)\n    y_val_tensor = torch.FloatTensor(y_val_seq).unsqueeze(1).to(device)\n    \n    # Create data loaders\n    train_dataset = TensorDataset(X_train_tensor, y_train_tensor)\n    val_dataset = TensorDataset(X_val_tensor, y_val_tensor)\n    train_loader = DataLoader(train_dataset, batch_size=32, shuffle=True)\n    val_loader = DataLoader(val_dataset, batch_size=32, shuffle=False)\n    \n    # Initialize model\n    model = LSTMModel(input_size=X_train.shape[1]).to(device)\n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(model.parameters(), lr=0.001)\n    \n    # Training loop\n    train_losses = []\n    val_losses = []\n    \n    for epoch in range(epochs):\n        model.train()\n        train_loss = 0\n        for batch_x, batch_y in train_loader:\n            optimizer.zero_grad()\n            outputs = model(batch_x)\n            loss = criterion(outputs, batch_y)\n            loss.backward()\n            optimizer.step()\n            train_loss += loss.item()\n        \n        model.eval()\n        val_loss = 0\n        with torch.no_grad():\n            for batch_x, batch_y in val_loader:\n                outputs = model(batch_x)\n                loss = criterion(outputs, batch_y)\n                val_loss += loss.item()\n        \n        train_losses.append(train_loss / len(train_loader))\n        val_losses.append(val_loss / len(val_loader))\n        \n        if epoch % 10 == 0:\n            print(f'Epoch {epoch}, Train Loss: {train_losses[-1]:.4f}, Val Loss: {val_losses[-1]:.4f}')\n    \n    return model, (train_losses, val_losses)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T08:40:04.116344Z","iopub.execute_input":"2025-07-21T08:40:04.116718Z","iopub.status.idle":"2025-07-21T08:40:04.136308Z","shell.execute_reply.started":"2025-07-21T08:40:04.116682Z","shell.execute_reply":"2025-07-21T08:40:04.135463Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class EnsembleModel:\n    \"\"\"Ensemble of multiple models with different strengths\"\"\"\n    \n    def __init__(self):\n        self.models = {}\n        self.weights = {}\n        self.scalers = {}\n        self.is_fitted = False\n    \n    def _initialize_models(self):\n        \"\"\"Initialize all models in the ensemble\"\"\"\n        self.models = {\n            'xgboost': xgb.XGBRegressor(\n                n_estimators=500,\n                max_depth=6,\n                learning_rate=0.01,\n                subsample=0.8,\n                colsample_bytree=0.8,\n                random_state=42,\n                n_jobs=-1\n            ),\n            'lightgbm': lgb.LGBMRegressor(\n                n_estimators=500,\n                max_depth=6,\n                learning_rate=0.01,\n                subsample=0.8,\n                colsample_bytree=0.8,\n                random_state=42,\n                n_jobs=-1,\n                verbose=-1\n            ),\n            # 'rf': RandomForestRegressor(\n            #     n_estimators=300,\n            #     max_depth=10,\n            #     min_samples_split=10,\n            #     min_samples_leaf=5,\n            #     random_state=42,\n            #     n_jobs=-1\n            # ),\n            'gbr': GradientBoostingRegressor(\n                n_estimators=200,\n                max_depth=6,\n                learning_rate=0.01,\n                subsample=0.8,\n                random_state=42\n            ),\n            'ridge': Ridge(alpha=1.0, random_state=42),\n            'elastic': ElasticNet(alpha=0.1, l1_ratio=0.5, random_state=42)\n        }\n        \n        # Initialize scalers for each model\n        for model_name in self.models.keys():\n            if model_name in ['ridge', 'elastic']:\n                self.scalers[model_name] = StandardScaler()\n            else:\n                self.scalers[model_name] = RobustScaler()\n\n    def fit(self, X, y, validation_split=0.2):\n        \"\"\"Train all models in the ensemble\"\"\"\n        print(\"\\n5. Training Ensemble Models...\")\n        \n        self._initialize_models()\n        \n        # Split data for validation\n        split_idx = int(len(X) * (1 - validation_split))\n        X_train, X_val = X[:split_idx], X[split_idx:]\n        y_train, y_val = y[:split_idx], y[split_idx:]\n        \n        # Train each model\n        val_scores = {}\n        \n        for model_name, model in self.models.items():\n            print(f\"Training {model_name}...\")\n            \n            # Scale features\n            X_train_scaled = self.scalers[model_name].fit_transform(X_train)\n            X_val_scaled = self.scalers[model_name].transform(X_val)\n            \n            # Train model\n            model.fit(X_train_scaled, y_train)\n            \n            # Validate\n            val_pred = model.predict(X_val_scaled)\n            val_score = pearsonr(y_val, val_pred)[0]\n            val_scores[model_name] = val_score\n            \n            print(f\"{model_name} validation correlation: {val_score:.4f}\")\n        \n        # Calculate ensemble weights based on validation performance\n        total_score = sum(max(0, score) for score in val_scores.values())\n        if total_score > 0:\n            self.weights = {name: max(0, score) / total_score for name, score in val_scores.items()}\n        else:\n            self.weights = {name: 1.0 / len(self.models) for name in self.models.keys()}\n        \n        print(f\"\\nEnsemble weights: {self.weights}\")\n        \n        # Train LSTM if available\n        # if PYTORCH_AVAILABLE:\n        #     try:\n        #         lstm_model, _ = train_lstm_model(X_train, y_train, X_val, y_val)\n        #         if lstm_model is not None:\n        #             self.models['lstm'] = lstm_model\n        #             # Add LSTM to weights with moderate weight\n        #             self.weights['lstm'] = 0.1\n        #             # Renormalize weights\n        #             total_weight = sum(self.weights.values())\n        #             self.weights = {k: v / total_weight for k, v in self.weights.items()}\n        #     except Exception as e:\n        #         print(f\"LSTM training failed: {e}\")\n        \n        self.is_fitted = True\n        return self\n\n    def predict(self, X):\n        \"\"\"Make predictions using the ensemble\"\"\"\n        if not self.is_fitted:\n            raise ValueError(\"Model must be fitted before making predictions\")\n        \n        predictions = {}\n        \n        for model_name, model in self.models.items():\n            # if model_name == 'lstm':\n            #     # Handle LSTM prediction separately\n            #     if PYTORCH_AVAILABLE:\n            #         try:\n            #             X_scaled = self.scalers['xgboost'].transform(X)  # Use XGBoost scaler\n            #             X_seq, _ = create_sequences(X_scaled, seq_length=30)\n            #             if len(X_seq) > 0:\n            #                 X_tensor = torch.FloatTensor(X_seq)\n            #                 model.eval()\n            #                 with torch.no_grad():\n            #                     lstm_pred = model(X_tensor).numpy().flatten()\n            #                 # Pad predictions to match input length\n            #                 padded_pred = np.full(len(X), lstm_pred[-1])\n            #                 padded_pred[-len(lstm_pred):] = lstm_pred\n            #                 predictions[model_name] = padded_pred\n            #             else:\n            #                 predictions[model_name] = np.zeros(len(X))\n            #         except Exception as e:\n            #             print(f\"LSTM prediction failed: {e}\")\n            #             predictions[model_name] = np.zeros(len(X))\n            # else:\n            X_scaled = self.scalers[model_name].transform(X)\n            predictions[model_name] = model.predict(X_scaled)\n        \n        # Weighted ensemble prediction\n        ensemble_pred = np.zeros(len(X))\n        for model_name, pred in predictions.items():\n            ensemble_pred += self.weights[model_name] * pred\n        \n        return ensemble_pred \n\n    def get_feature_importance(self):\n        \"\"\"Get feature importance from tree-based models\"\"\"\n        importance_dict = {}\n        \n        for model_name, model in self.models.items():\n            if hasattr(model, 'feature_importances_'):\n                importance_dict[model_name] = model.feature_importances_\n        \n        return importance_dict","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T08:40:04.137379Z","iopub.execute_input":"2025-07-21T08:40:04.137702Z","iopub.status.idle":"2025-07-21T08:40:04.167045Z","shell.execute_reply.started":"2025-07-21T08:40:04.137672Z","shell.execute_reply":"2025-07-21T08:40:04.166018Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def evaluate_model(model, X, y, cv_folds=5):\n    \"\"\"Evaluate model using time series cross-validation\"\"\"\n    print(\"\\n6. Model Evaluation...\")\n    \n    tscv = TimeSeriesSplit(n_splits=cv_folds)\n    correlations = []\n    \n    for fold, (train_idx, val_idx) in enumerate(tscv.split(X)):\n        X_train_fold, X_val_fold = X[train_idx], X[val_idx]\n        y_train_fold, y_val_fold = y[train_idx], y[val_idx]\n        \n        # Create and train model for this fold\n        fold_model = EnsembleModel()\n        fold_model.fit(X_train_fold, y_train_fold, validation_split=0.2)\n        \n        # Predict\n        y_pred = fold_model.predict(X_val_fold)\n        \n        # Calculate correlation\n        correlation = pearsonr(y_val_fold, y_pred)[0]\n        correlations.append(correlation)\n        \n        print(f\"Fold {fold + 1} correlation: {correlation:.4f}\")\n    \n    mean_correlation = np.mean(correlations)\n    std_correlation = np.std(correlations)\n    \n    print(f\"\\nCross-validation results:\")\n    print(f\"Mean correlation: {mean_correlation:.4f} ± {std_correlation:.4f}\")\n    \n    return mean_correlation, std_correlation","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T08:40:04.167988Z","iopub.execute_input":"2025-07-21T08:40:04.168333Z","iopub.status.idle":"2025-07-21T08:40:04.195937Z","shell.execute_reply.started":"2025-07-21T08:40:04.168309Z","shell.execute_reply":"2025-07-21T08:40:04.194931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def main():\n    \"\"\"Main execution pipeline\"\"\"\n    print(\"Starting Crypto Price Prediction Pipeline...\")\n    \n    # Load data\n    train_df, test_df, feature_cols, public_cols = load_and_explore_data()\n    \n    # Preprocess data\n    train_df, test_df, all_feature_cols = preprocess_data(train_df, test_df, feature_cols, public_cols)\n    \n    # Prepare features and target\n    X = train_df[all_feature_cols].values\n    y = train_df['label'].values\n    X_test = test_df[all_feature_cols].values\n    \n    # Feature selection\n    X_selected, selected_features, selector = perform_feature_selection(\n        pd.DataFrame(X, columns=all_feature_cols), y, method='mutual_info', k=300\n    )\n    \n    # Apply same selection to test data\n    X_test_selected = selector.transform(pd.DataFrame(X_test, columns=all_feature_cols))\n    \n    print(f\"\\nUsing {len(selected_features)} selected features\")\n    \n    # Focus on recent data as suggested in the tips\n    recent_months = 2  # Use last 6 months\n    cutoff_date = train_df.index.max() - pd.DateOffset(months=recent_months)\n    recent_mask = train_df.index >= cutoff_date\n    \n    X_recent = X_selected[recent_mask]\n    y_recent = y[recent_mask]\n    \n    print(f\"Using recent {recent_months} months of data: {len(X_recent)} samples\")\n    \n    # Train final ensemble model\n    final_model = EnsembleModel()\n    final_model.fit(X_recent, y_recent, validation_split=0.2)\n    \n    # Make predictions on test set\n    test_predictions = final_model.predict(X_test_selected)\n    \n    # Create submission file\n    submission = pd.DataFrame({\n        'ID': test_df['timestamp'] if 'timestamp' in test_df.columns else range(len(test_df)),\n        'label': test_predictions\n    })\n    \n    submission.to_csv('/kaggle/working/submission.csv', index=False)\n    print(f\"\\nSubmission file created: submission.csv\")\n    print(f\"Test predictions statistics:\")\n    print(f\"Mean: {test_predictions.mean():.6f}\")\n    print(f\"Std: {test_predictions.std():.6f}\")\n    print(f\"Min: {test_predictions.min():.6f}\")\n    print(f\"Max: {test_predictions.max():.6f}\")\n    \n    # Feature importance analysis\n    importance_dict = final_model.get_feature_importance()\n    if importance_dict:\n        print(f\"\\nTop 10 most important features:\")\n        for model_name, importance in importance_dict.items():\n            if len(importance) > 0:\n                top_indices = np.argsort(importance)[-10:][::-1]\n                print(f\"\\n{model_name}:\")\n                for i, idx in enumerate(top_indices):\n                    feature_name = selected_features[idx]\n                    print(f\"  {i+1}. {feature_name}: {importance[idx]:.4f}\")\n    \n    return final_model, submission\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T08:40:04.196923Z","iopub.execute_input":"2025-07-21T08:40:04.197190Z","iopub.status.idle":"2025-07-21T08:40:04.225297Z","shell.execute_reply.started":"2025-07-21T08:40:04.197168Z","shell.execute_reply":"2025-07-21T08:40:04.224286Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if __name__ == '__main__':\n    model, submission = main()\n    print(\"\\n\" + \"=\"*50)\n    print(\"Pipeline completed successfully!\")\n    print(\"=\"*50)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T09:07:41.376237Z","iopub.status.idle":"2025-07-21T09:07:41.376504Z","shell.execute_reply.started":"2025-07-21T09:07:41.376383Z","shell.execute_reply":"2025-07-21T09:07:41.376393Z"}},"outputs":[],"execution_count":null}]}