{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84493,"databundleVersionId":11305158,"sourceType":"competition"}],"dockerImageVersionId":30919,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":13703.539,"end_time":"2025-03-04T00:22:18.733470","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-03-03T20:33:55.194470","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Jane Street Market Prediction","metadata":{"_uuid":"b5114445-88d8-4500-8429-5d64a26f972b","_cell_guid":"cb302c2b-8257-4785-bb60-4a74385b03ab","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"markdown","source":"# IMPORTS & CONFIGURATION & UTILITIES","metadata":{"_uuid":"ec37c7b5-d0cb-4530-82b1-8ec5fa358f6f","_cell_guid":"29a547b9-cca4-4cba-8669-60b2a822dd38","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport lightgbm as lgb\nfrom sklearn.model_selection import TimeSeriesSplit\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import mean_squared_error\nimport warnings\nimport os\nimport multiprocessing\nfrom numba import njit, prange\nimport gc\n\n\n# Disable warnings and set performance-related settings\nwarnings.filterwarnings('ignore')\npd.options.mode.chained_assignment = None  # Suppress SettingWithCopyWarning\nos.environ['NUMEXPR_MAX_THREADS'] = str(multiprocessing.cpu_count())\n\n# Global configuration\nCONFIG = {\n    'use_gpu': True,           # Set to False if GPU not available\n    'sample_size': 200000,       # Set to an integer for sampling\n    'max_partitions': 10,      # Number of data partitions to load\n    'use_dask': True,         # Set to True for distributed computing\n    'target_column': 'responder_6',  # Target variable\n    'weight_column': 'weight', # Weight column\n    'random_seed': 42,         # Random seed for reproducibility\n    'data_path': '/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet', \n    'memory_efficient': True,  # Enable memory optimizations\n    'verbose': False           # Enable detailed logging\n}\n\ndef log(message, force=False):\n    \"\"\"Utility function for controlled logging\"\"\"\n    if CONFIG['verbose'] or force:\n        print(message)","metadata":{"_uuid":"78113b0f-605b-46c7-9e37-1200ecb9c6f0","_cell_guid":"a446295b-4c7c-4da2-bae4-b28c252c7b55","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# DATA LOADING FUNCTIONS","metadata":{"_uuid":"6c31cc41-2a54-41e0-83bc-7e7c383d62ed","_cell_guid":"8343132f-fe4c-4851-b1f3-7516d18a44e7","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"def get_optimal_dtypes(df):\n    \"\"\"\n    Optimize dataframe memory usage by selecting appropriate dtypes\n    \n    Parameters:\n    df (pandas.DataFrame): Input dataframe\n    \n    Returns:\n    dict: Optimized dtypes for columns\n    \"\"\"\n    dtypes = {}\n    \n    for col in df.columns:\n        # Categoricals\n        if col in ['symbol_id'] or df[col].nunique() < 100:\n            dtypes[col] = 'category'\n        # Integers\n        elif 'id' in col or col.endswith('_id'):\n            dtypes[col] = 'int32'\n        # Floats\n        elif df[col].dtype == np.float64:\n            dtypes[col] = 'float32'\n    \n    return dtypes\n\ndef memory_efficient_load_data(file_path, sample_size=None):\n    \"\"\"\n    Load data with minimal memory usage\n    \n    Parameters:\n    file_path (str): Path to the parquet file\n    sample_size (int, optional): Number of rows to sample\n    \n    Returns:\n    pandas.DataFrame: Loaded and optimized dataframe\n    \"\"\"\n    # First load a small sample to determine dtypes\n    sample_df = pd.read_parquet(file_path, engine='pyarrow')\n    if sample_size is not None:\n        sample_df = sample_df.sample(min(100000, sample_size), random_state=CONFIG['random_seed'])\n    \n    # Get optimized dtypes\n    dtypes = get_optimal_dtypes(sample_df)\n    del sample_df\n    gc.collect()\n    \n    # Read parquet without dtypes\n    df = pd.read_parquet(file_path, engine='pyarrow')\n    \n    # Sample if needed\n    if sample_size is not None and sample_size < len(df):\n        df = df.sample(sample_size, random_state=CONFIG['random_seed'])\n    \n    # Apply optimized dtypes\n    for col, dtype in dtypes.items():\n        df[col] = df[col].astype(dtype)\n    \n    # Force garbage collection\n    gc.collect()\n    \n    return df\ndef load_train_data(sample_size=None, max_partitions=10, use_dask=False):\n    \"\"\"\n    Optimized data loading with optional Dask support for large datasets\n    \n    Parameters:\n    sample_size (int, optional): Number of rows to sample from each partition\n    max_partitions (int): Maximum number of partitions to load\n    use_dask (bool): Use Dask for distributed loading\n    \n    Returns:\n    pandas.DataFrame: Combined training data\n    \"\"\"\n    log(\"Loading training data...\", force=True)\n    \n    if use_dask:\n        try:\n            import dask.dataframe as dd\n            dfs = []\n            for partition_id in range(max_partitions):\n                partition_path = f'{CONFIG[\"data_path\"]}/partition_id={partition_id}/part-0.parquet'\n                try:\n                    ddf = dd.read_parquet(partition_path)\n                    if sample_size:\n                        ddf = ddf.sample(frac=sample_size/len(ddf))\n                    dfs.append(ddf)\n                    log(f\"Loaded partition {partition_id}\")\n                except Exception as e:\n                    log(f\"Error loading partition {partition_id}: {e}\")\n            \n            log(\"Computing combined dataframe...\")\n            combined_df = dd.concat(dfs).compute()\n            \n            # Apply memory optimizations\n            if CONFIG['memory_efficient']:\n                dtypes = get_optimal_dtypes(combined_df)\n                for col, dtype in dtypes.items():\n                    combined_df[col] = combined_df[col].astype(dtype)\n            \n            log(f\"Combined data shape: {combined_df.shape}\", force=True)\n            return combined_df\n        \n        except ImportError:\n            log(\"Dask not available, falling back to pandas\")\n    \n    # Pandas-based loading\n    train_dfs = []\n    \n    for partition_id in range(max_partitions):\n        partition_path = f'{CONFIG[\"data_path\"]}/partition_id={partition_id}/part-0.parquet'\n        \n        try:\n            if CONFIG['memory_efficient']:\n                df = memory_efficient_load_data(partition_path, sample_size)\n            else:\n                df = pd.read_parquet(partition_path, engine='pyarrow')\n                if sample_size:\n                    df = df.sample(sample_size, random_state=CONFIG['random_seed'])\n            \n            train_dfs.append(df)\n            log(f\"Loaded partition {partition_id} with shape {df.shape}\")\n            \n            # Free memory\n            gc.collect()\n            \n        except Exception as e:\n            log(f\"Error loading partition {partition_id}: {e}\")\n    \n    # Combine all partitions\n    combined_df = pd.concat(train_dfs, ignore_index=True)\n    log(f\"Combined training data shape: {combined_df.shape}\", force=True)\n    \n    # Clear memory\n    del train_dfs\n    gc.collect()\n    \n    return combined_df","metadata":{"_uuid":"f1a73251-03af-47fe-862f-224c6a0ada8b","_cell_guid":"6bcad5be-4f74-44d9-8f49-d88f3eeb1cbd","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# FEATURE ENGINEERING","metadata":{"_uuid":"bc629250-ba14-40fc-a69e-4ea85f8c920a","_cell_guid":"00311797-5997-4ef4-9875-8c352062bcae","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"@njit\ndef compute_rolling_features(col_data, window_sizes):\n    \"\"\"\n    Compute rolling features for a single column\n    \n    Parameters:\n    col_data (np.ndarray): Input column data\n    window_sizes (tuple): Window sizes for rolling computations\n    \n    Returns:\n    np.ndarray: Computed rolling features\n    \"\"\"\n    rows = len(col_data)\n    result = np.zeros((rows, len(window_sizes) * 2))\n    \n    for idx, window in enumerate(window_sizes):\n        # Rolling mean\n        rolling_mean = np.zeros_like(col_data, dtype=np.float32)\n        for i in range(rows):\n            start = max(0, i - window + 1)\n            if i >= start:\n                rolling_mean[i] = np.mean(col_data[start:i+1])\n        \n        # Rolling standard deviation\n        rolling_std = np.zeros_like(col_data, dtype=np.float32)\n        for i in range(rows):\n            start = max(0, i - window + 1)\n            if i >= start and i - start > 1:  # Need at least 2 points for std\n                rolling_std[i] = np.std(col_data[start:i+1])\n        \n        result[:, idx * 2] = rolling_mean\n        result[:, idx * 2 + 1] = rolling_std\n    \n    return result\n\n@njit(parallel=True)\ndef fast_rolling_features(data, window_sizes=(3, 5)):\n    \"\"\"\n    Numba-accelerated rolling window feature computation\n    \n    Parameters:\n    data (np.ndarray): Input data array\n    window_sizes (tuple): Window sizes for rolling computations\n    \n    Returns:\n    np.ndarray: Computed rolling features\n    \"\"\"\n    rows, cols = data.shape\n    \n    # Preallocate result array\n    result = np.zeros((rows, cols * len(window_sizes) * 2), dtype=np.float32)\n    \n    # Compute rolling features for each column\n    for col in prange(cols):\n        col_features = compute_rolling_features(data[:, col], window_sizes)\n        \n        # Copy computed features to result array\n        for w_idx in range(len(window_sizes)):\n            result[:, col * len(window_sizes) * 2 + w_idx * 2] = col_features[:, w_idx * 2]\n            result[:, col * len(window_sizes) * 2 + w_idx * 2 + 1] = col_features[:, w_idx * 2 + 1]\n    \n    return result\n\ndef advanced_feature_engineering(df, memory_efficient=True):\n    \"\"\"\n    Optimized feature engineering with Numba and vectorized operations\n    \n    Parameters:\n    df (pandas.DataFrame): Input dataframe\n    memory_efficient (bool): Use memory-efficient operations\n    \n    Returns:\n    pandas.DataFrame: Dataframe with engineered features\n    \"\"\"\n    log(\"Performing feature engineering...\", force=True)\n    \n    # Create a copy or work with input dataframe\n    if memory_efficient:\n        result = df  # Work with original to save memory\n    else:\n        result = df.copy()\n    \n    # Identify feature columns\n    feature_cols = [col for col in result.columns if col.startswith('feature_')]\n    log(f\"Found {len(feature_cols)} feature columns\")\n    \n    # 1. Time-based Features (Vectorized)\n    result['day_of_week'] = result['date_id'] % 7\n    result['time_of_day'] = result['time_id'] % 24\n    \n    # 2. Rolling Window Features (Vectorized + Numba)\n    # Process in chunks to manage memory\n    chunk_size = 10 if memory_efficient else len(feature_cols)\n    \n    for i in range(0, len(feature_cols), chunk_size):\n        chunk_cols = feature_cols[i:i+chunk_size]\n        log(f\"Processing chunk {i//chunk_size + 1}/{(len(feature_cols)-1)//chunk_size + 1}\")\n        \n        feature_matrix = result[chunk_cols].values\n        rolling_features = fast_rolling_features(feature_matrix)\n        \n        # Add rolling features back to dataframe\n        for j, col in enumerate(chunk_cols):\n            for k, suffix in enumerate(['_roll_mean_3', '_roll_std_3', '_roll_mean_5', '_roll_std_5']):\n                result[f'{col}{suffix}'] = rolling_features[:, j * 4 + k]\n        \n        # Clean up to free memory\n        del feature_matrix, rolling_features\n        gc.collect()\n    \n    # 3. Focus on top features for other transformations to save memory\n    top_features = feature_cols[:min(20, len(feature_cols))]\n    \n    # 4. Momentum Features (Vectorized)\n    for feature in top_features[:min(10, len(top_features))]:\n        result[f'{feature}_momentum'] = result.groupby('symbol_id')[feature].pct_change()\n    \n    # 5. Cross-sectional Z-score (Vectorized)\n    for feature in top_features[:min(5, len(top_features))]:\n        result[f'{feature}_zscore'] = result.groupby('date_id')[feature].transform(\n            lambda x: (x - x.mean()) / (x.std() + 1e-8)\n        )\n    \n    # 6. Feature Interactions (Vectorized)\n    for i, feat1 in enumerate(top_features[:3]):\n        for feat2 in top_features[i+1:i+3]:  # Limit interactions to save memory\n            # Multiplicative interactions\n            result[f'{feat1}_{feat2}_interaction'] = result[feat1] * result[feat2]\n    \n    # 7. Target Encoding (Vectorized)\n    result['symbol_target_mean'] = result.groupby('symbol_id')[CONFIG['target_column']].transform('mean')\n    result['day_target_mean'] = result.groupby('day_of_week')[CONFIG['target_column']].transform('mean')\n    \n    # Fill remaining NaNs\n    result = result.fillna(0)\n    \n    log(\"Feature engineering completed\")\n    log(f\"Final dataframe shape: {result.shape}\", force=True)\n    \n    return result","metadata":{"_uuid":"1fd05775-f0f7-472f-ac3a-8958c20b9958","_cell_guid":"1479327d-c308-4f04-8aa7-86c1423544ae","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# MODEL TRAINING","metadata":{"_uuid":"156cd9ac-8a9a-4d18-a9e7-c283648c2c74","_cell_guid":"9a436f44-aac6-4135-8195-b895885524d2","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"def get_lgb_params(use_gpu=False):\n    \"\"\"\n    Get optimized LightGBM parameters\n    \n    Parameters:\n    use_gpu (bool): Enable GPU acceleration\n    \n    Returns:\n    dict: LightGBM parameters\n    \"\"\"\n    params = {\n        'objective': 'regression',\n        'metric': 'rmse',\n        'boosting_type': 'dart',  # More robust to overfitting\n        'learning_rate': 0.05,\n        'num_leaves': 128,\n        'max_depth': 10,\n        'subsample': 0.8,\n        'colsample_bytree': 0.8,\n        'n_estimators': 3000,\n        'early_stopping_rounds': 150,\n        'min_child_samples': 30,\n        'reg_alpha': 0.2,\n        'reg_lambda': 0.2,\n        'n_jobs': -1,  # Use all available cores\n        'verbose': -1\n    }\n    \n    # Add GPU parameters if enabled\n    if use_gpu:\n        try:\n            params['device'] = 'gpu'\n            params['gpu_platform_id'] = 0\n            params['gpu_device_id'] = 0\n        except:\n            log(\"GPU acceleration not available, falling back to CPU\")\n    \n    return params\n\ndef train_enhanced_model(train_df, feature_cols, target_col='responder_6', weight_col='weight'):\n    \"\"\"\n    Optimized LightGBM model training with advanced configurations\n    \n    Parameters:\n    train_df (pandas.DataFrame): Training dataframe\n    feature_cols (list): List of feature columns\n    target_col (str): Target column name\n    weight_col (str): Weight column name\n    \n    Returns:\n    tuple: (model, validation_metrics)\n    \"\"\"\n    log(\"Training model...\", force=True)\n    \n    # Get model parameters\n    params = get_lgb_params(CONFIG['use_gpu'])\n    \n    # Time-based train-validation split\n    date_ids = sorted(train_df['date_id'].unique())\n    split_idx = int(len(date_ids) * 0.8)\n    \n    train_dates = date_ids[:split_idx]\n    valid_dates = date_ids[split_idx:]\n    \n    train_mask = train_df['date_id'].isin(train_dates)\n    valid_mask = train_df['date_id'].isin(valid_dates)\n    \n    X_train = train_df.loc[train_mask, feature_cols]\n    y_train = train_df.loc[train_mask, target_col]\n    w_train = train_df.loc[train_mask, weight_col]\n    \n    X_valid = train_df.loc[valid_mask, feature_cols]\n    y_valid = train_df.loc[valid_mask, target_col]\n    w_valid = train_df.loc[valid_mask, weight_col]\n    \n    log(f\"Training set: {X_train.shape}, Validation set: {X_valid.shape}\")\n    \n    # Create LightGBM datasets with categorical feature support\n    categorical_features = [\n        col for col in feature_cols \n        if train_df[col].dtype.name == 'category' or \n        (train_df[col].nunique() < 100 and 'float' not in str(train_df[col].dtype))\n    ]\n    \n    train_dataset = lgb.Dataset(\n        X_train, \n        y_train, \n        weight=w_train, \n        categorical_feature=categorical_features\n    )\n    valid_dataset = lgb.Dataset(\n        X_valid, \n        y_valid, \n        weight=w_valid, \n        categorical_feature=categorical_features\n    )\n    \n    # Train model\n    model = lgb.train(\n        params,\n        train_dataset,\n        valid_sets=[train_dataset, valid_dataset],\n        callbacks=[\n            lgb.early_stopping(150),\n            lgb.log_evaluation(100)\n        ]\n    )\n    \n    # Predictions and metrics\n    y_pred = model.predict(X_valid)\n    \n    # Custom weighted metrics\n    mse = mean_squared_error(y_valid, y_pred, sample_weight=w_valid)\n    weighted_r2 = 1 - (np.sum(w_valid * (y_valid - y_pred)**2) / \n                       np.sum(w_valid * (y_valid - y_valid.mean())**2))\n    \n    log(f\"Model training completed. Best iteration: {model.best_iteration}\", force=True)\n    log(f\"MSE: {mse:.6f}, Weighted R²: {weighted_r2:.6f}\", force=True)\n    \n    # Free memory\n    del X_train, X_valid, y_train, y_valid, w_train, w_valid\n    gc.collect()\n    \n    return model, {\n        'mse': mse,\n        'weighted_r2': weighted_r2,\n        'best_iteration': model.best_iteration\n    }","metadata":{"_uuid":"9b040c73-cdbc-4667-8367-3bc7d856eba7","_cell_guid":"3d823d6d-07d1-4723-a3ae-61f0f675f022","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# PREDICTION FUNCTIONS","metadata":{"_uuid":"c1608652-6ef7-4e74-be79-425c0f18d7d3","_cell_guid":"2f4e2ea8-7589-418e-9a8a-3c448e2b4263","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"def predict_batch(model, data, feature_cols, batch_size=100000):\n    \"\"\"\n    Make predictions in batches to manage memory\n    \n    Parameters:\n    model (LGBMModel): Trained LightGBM model\n    data (pandas.DataFrame): Data to predict on\n    feature_cols (list): Feature columns\n    batch_size (int): Batch size for prediction\n    \n    Returns:\n    numpy.ndarray: Predictions\n    \"\"\"\n    n_samples = len(data)\n    predictions = np.zeros(n_samples, dtype=np.float32)\n    \n    for i in range(0, n_samples, batch_size):\n        end = min(i + batch_size, n_samples)\n        batch_data = data.iloc[i:end]\n        batch_preds = model.predict(batch_data[feature_cols])\n        predictions[i:end] = batch_preds\n    \n    return predictions","metadata":{"_uuid":"33f65f82-4a9c-4ee9-8303-536f58c568a0","_cell_guid":"84c7c418-d654-4e00-bf6a-34d7e06bb767","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# MAIN EXECUTION FUNCTIONS","metadata":{"_uuid":"c83a2a89-9ca4-4e8e-9396-719723505b2a","_cell_guid":"fdbb3157-4041-43c4-9079-b849ff8b08d7","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"def prepare_data():\n    \"\"\"\n    Load and prepare data\n    \n    Returns:\n    pandas.DataFrame: Prepared dataframe\n    \"\"\"\n    # Load training data\n    train_df = load_train_data(\n        sample_size=CONFIG['sample_size'],\n        max_partitions=CONFIG['max_partitions'],\n        use_dask=CONFIG['use_dask']\n    )\n    \n    # Basic exploratory analysis\n    log(\"\\n📊 Basic Data Overview:\", force=True)\n    log(f\"Total Rows: {len(train_df)}\", force=True)\n    log(f\"Unique Symbols: {train_df['symbol_id'].nunique()}\", force=True)\n    log(f\"Date Range: {train_df['date_id'].min()} - {train_df['date_id'].max()}\", force=True)\n    \n    return train_df\n\ndef remove_correlated_features(train_df, feature_cols, threshold=0.9):\n    \"\"\"\n    Remove highly correlated features based on a threshold.\n    \n    Parameters:\n    train_df (pd.DataFrame): The training dataframe.\n    feature_cols (list): List of feature columns to consider.\n    threshold (float): Correlation threshold (default: 0.9).\n    \n    Returns:\n    list: List of selected features after removing correlated ones.\n    \"\"\"\n    # Compute the correlation matrix\n    corr_matrix = train_df[feature_cols].corr().abs()\n    \n    # Select upper triangle of correlation matrix\n    upper_tri = corr_matrix.where(np.triu(np.ones(corr_matrix.shape), k=1).astype(bool))\n    \n    # Find features to drop\n    to_drop = [column for column in upper_tri.columns if any(upper_tri[column] > threshold)]\n    \n    # Keep only the features that are not highly correlated\n    selected_features = [col for col in feature_cols if col not in to_drop]\n    \n    log(f\"Removed {len(to_drop)} highly correlated features\", force=True)\n    log(f\"Selected {len(selected_features)} features after correlation filtering\", force=True)\n    \n    return selected_features\n\ndef engineer_features(train_df):\n    \"\"\"\n    Perform feature engineering and correlation-based feature selection.\n    \"\"\"\n    # Feature engineering\n    train_df_engineered = advanced_feature_engineering(\n        train_df, \n        memory_efficient=CONFIG['memory_efficient']\n    )\n    \n    # Feature selection with comprehensive filtering\n    feature_cols = [\n        col for col in train_df_engineered.columns \n        if (col.startswith('feature_') or \n            any(suffix in col for suffix in [\n                '_roll_mean', '_roll_std', \n                '_momentum', '_zscore', \n                '_interaction', '_ratio', \n                '_target_mean', 'day_of_week', 'time_of_day'\n            ])) \n        and col not in [CONFIG['target_column'], CONFIG['weight_column']]\n    ]\n    \n    log(f\"Selected {len(feature_cols)} features\", force=True)\n    \n    # Remove highly correlated features\n    selected_features = remove_correlated_features(train_df_engineered, feature_cols, threshold=0.9)\n    \n    log(f\"Remaining {len(selected_features)} features after correlation filtering\", force=True)\n    \n    return train_df_engineered, selected_features  \n\n\ndef train_model(train_df_engineered, feature_cols):\n    \"\"\"\n    Train prediction model with correlation-based feature selection.\n    \"\"\"\n    # Train model with selected features\n    model, metrics = train_enhanced_model(\n        train_df_engineered, \n        feature_cols,\n        target_col=CONFIG['target_column'],\n        weight_col=CONFIG['weight_column']\n    )\n    \n    # Print results\n    log(\"\\n📈 Model Performance with Selected Features:\", force=True)\n    log(f\"Mean Squared Error: {metrics['mse']:.6f}\", force=True)\n    log(f\"Weighted R-squared: {metrics['weighted_r2']:.6f}\", force=True)\n    log(f\"Best Iteration: {metrics['best_iteration']}\", force=True)\n    \n    return model, metrics, feature_cols\n\n\ndef save_model(model, feature_cols, output_path='model.txt'):\n    \"\"\"\n    Save trained model and feature columns\n    \n    Parameters:\n    model (LGBMModel): Trained model\n    feature_cols (list): Feature columns\n    output_path (str): Output path for model\n    \"\"\"\n    # Save model\n    model.save_model(output_path)\n    log(f\"Model saved to {output_path}\", force=True)\n    \n    # Save feature columns\n    import json\n    with open('feature_cols.json', 'w') as f:\n        json.dump(feature_cols, f)\n    log(\"Feature columns saved to feature_cols.json\", force=True)","metadata":{"_uuid":"b34e59e8-39e4-4db1-b281-9587d7c04fe1","_cell_guid":"eb8f0a4c-e4b2-4fd7-922d-0f4c2e3363e2","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# MAIN FUNCTION","metadata":{"_uuid":"e7adba98-c0d5-498d-9be2-6f2d1a85f925","_cell_guid":"439d9400-c185-458b-8980-0795b42456e8","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"def main():\n    \"\"\"\n    Main execution function with correlation-based feature selection.\n    \"\"\"\n    # Load and prepare data\n    train_df = prepare_data()\n    \n    # Feature engineering and correlation-based feature selection\n    train_df_engineered, selected_features = engineer_features(train_df)\n    \n    # Free memory\n    del train_df\n    gc.collect()\n    \n    # Train model with selected features\n    model, metrics, selected_features = train_model(train_df_engineered, selected_features)\n    \n    # Save model and selected features\n    save_model(model, selected_features)\n    \n    return model, metrics, selected_features","metadata":{"_uuid":"7abc855d-7eb4-460f-a9dd-4b453d85073d","_cell_guid":"707d4521-3ac2-470b-9a94-b59fc1b1818f","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# STANDALONE EXECUTION","metadata":{"_uuid":"23de2fe0-cdae-464c-92cf-fa7a54719061","_cell_guid":"16eed308-0e7b-4759-96c6-dedd12355012","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"if __name__ == \"__main__\":\n    # Set configuration options here\n    CONFIG.update({\n        'use_gpu': True,\n        'sample_size': 200000,\n        'max_partitions': 5,\n        'use_dask': True,\n        'memory_efficient': False,\n        'verbose': True\n    })\n    \n    # Run full pipeline\n    model, metrics, selected_feature_cols = main()","metadata":{"_uuid":"450a554f-8e79-45a6-9c67-b6324b738034","_cell_guid":"9cb67980-f720-4870-8c4b-2454589a0ff8","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null}]}