{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import KFold, TimeSeriesSplit\nfrom sklearn.preprocessing import StandardScaler, RobustScaler\nfrom sklearn.feature_selection import mutual_info_regression\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nfrom catboost import CatBoostRegressor\nfrom scipy.stats import pearsonr, spearmanr\nfrom scipy.special import expit  # Sigmoid function\nimport warnings\nwarnings.filterwarnings('ignore')\n\nclass Config:\n    \"\"\"Configuration parameters for memory-optimized prediction pipeline\"\"\"\n    # Paths\n    train_path = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    test_path = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    sample_sub_path = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n    \n    # Optimized settings for memory efficiency\n    use_recent_months = 9  \n    n_top_features = 150   # Final features to select\n    n_folds = 5\n    random_seed = 42\n    stratified_sample_size = 100000  # Reduced from 150k\n    \n    # Feature selection parameters\n    correlation_threshold = 0.95  # Stricter to remove more correlated features\n    max_complex_features = 100  # Limit complex transformations\n    \n    # Feature creation limits\n    max_base_features_for_complex = 30  # Reduced from 50\n    max_polynomial_features = 20  # Limit polynomial interactions\n    max_rolling_windows = 3  # Reduced from 5\n\ndef load_and_prepare_data():\n    \"\"\"Load and prepare data with enhanced preprocessing\"\"\"\n    print(\"Loading data...\")\n    train = pd.read_parquet(Config.train_path)\n    test = pd.read_parquet(Config.test_path)\n    \n    # Handle timestamp-based filtering if applicable\n    if Config.use_recent_months and 'timestamp' in train.columns:\n        train['timestamp'] = pd.to_datetime(train['timestamp'])\n        cutoff = train['timestamp'].max() - pd.DateOffset(months=Config.use_recent_months)\n        train = train[train['timestamp'] >= cutoff].reset_index(drop=True)\n        print(f\"Using data from last {Config.use_recent_months} months: {len(train)} rows\")\n    \n    return train, test\n\ndef get_time_weighted_sample(df, sample_size=100000):\n    \"\"\"Create a time-weighted sample that favors recent data\"\"\"\n    n_rows = len(df)\n    \n    # Create exponential weights that favor recent data\n    # Weight increases exponentially towards the end of the dataset\n    positions = np.arange(n_rows)\n    weights = np.exp(positions / n_rows * 2)  # Exponential growth\n    weights = weights / weights.sum()  # Normalize\n    \n    # Sample with replacement using weights\n    sample_indices = np.random.choice(\n        n_rows, \n        size=sample_size, \n        replace=True, \n        p=weights\n    )\n    \n    # Get unique indices and sort them\n    unique_indices = np.unique(sample_indices)\n    \n    print(f\"Time-weighted sample: {len(unique_indices)} unique rows from {sample_size} samples\")\n    print(f\"Sample distribution - First 25%: {(unique_indices < n_rows*0.25).sum()}, \"\n          f\"Last 25%: {(unique_indices >= n_rows*0.75).sum()}\")\n    \n    return df.iloc[unique_indices].copy()\n\ndef remove_correlated_features(df, features, threshold=0.95):\n    \"\"\"Remove highly correlated features to reduce multicollinearity\"\"\"\n    print(f\"Removing correlated features (threshold: {threshold})...\")\n    \n    # Work with a sample for correlation calculation to save memory\n    sample_df = get_time_weighted_sample(df, sample_size=50000)\n    \n    # Calculate correlation matrix on sample\n    corr_matrix = sample_df[features].corr().abs()\n    upper_triangle = corr_matrix.where(\n        np.triu(np.ones(corr_matrix.shape), k=1).astype(bool)\n    )\n    \n    # Find features to drop\n    to_drop = [column for column in upper_triangle.columns \n               if any(upper_triangle[column] > threshold)]\n    \n    print(f\"Removed {len(to_drop)} highly correlated features\")\n    return [f for f in features if f not in to_drop]\n\ndef create_selective_complex_transformations(df, base_features, max_features=100):\n    \"\"\"Create selective complex transformations with memory efficiency\"\"\"\n    print(\"Creating selective complex feature transformations...\")\n    transformed_features = []\n    \n    # First, identify most important base features using a quick sample\n    sample_df = get_time_weighted_sample(df, sample_size=20000)\n    feature_importance = []\n    \n    for feat in base_features[:50]:  # Check top 50 base features\n        if feat in sample_df.columns and 'label' in sample_df.columns:\n            corr = abs(np.corrcoef(sample_df[feat].fillna(0), sample_df['label'])[0, 1])\n            feature_importance.append((feat, corr))\n    \n    # Sort by importance and select top features\n    feature_importance.sort(key=lambda x: x[1], reverse=True)\n    important_features = [feat for feat, _ in feature_importance[:Config.max_base_features_for_complex]]\n    \n    print(f\"  Selected {len(important_features)} most important features for transformations\")\n    \n    # 1. Limited polynomial features (only top features)\n    print(\"  - Creating limited polynomial features...\")\n    poly_count = 0\n    for i, feat1 in enumerate(important_features[:10]):\n        if poly_count >= Config.max_polynomial_features:\n            break\n        for j, feat2 in enumerate(important_features[i:i+3]):\n            if poly_count >= Config.max_polynomial_features:\n                break\n            if feat1 in df.columns and feat2 in df.columns:\n                poly_name = f'poly_{feat1}_{feat2}'\n                df[poly_name] = df[feat1] * df[feat2]\n                transformed_features.append(poly_name)\n                poly_count += 1\n    \n    # 2. Selective exponential and logarithmic transformations\n    print(\"  - Creating selective exponential and logarithmic features...\")\n    for feat in important_features[:15]:  # Only top 15\n        if feat in df.columns:\n            # Log transform only\n            log_name = f'{feat}_log'\n            min_val = df[feat].min()\n            offset = abs(min_val) + 1 if min_val <= 0 else 0\n            df[log_name] = np.log(df[feat] + offset + 1e-10)\n            transformed_features.append(log_name)\n            \n            # Sigmoid for top 10 only\n            if feat in important_features[:10]:\n                sigmoid_name = f'{feat}_sigmoid'\n                df[sigmoid_name] = expit(df[feat] / (df[feat].std() + 1e-10))\n                transformed_features.append(sigmoid_name)\n    \n    # 3. Limited trigonometric transformations\n    print(\"  - Creating limited trigonometric features...\")\n    for feat in important_features[:8]:  # Only top 8\n        if feat in df.columns:\n            normalized = 2 * np.pi * (df[feat] - df[feat].min()) / (df[feat].max() - df[feat].min() + 1e-10)\n            sin_name = f'{feat}_sin'\n            df[sin_name] = np.sin(normalized)\n            transformed_features.append(sin_name)\n    \n    # 4. Single RBF transformation per feature\n    print(\"  - Creating limited RBF features...\")\n    for feat in important_features[:10]:  # Only top 10\n        if feat in df.columns:\n            rbf_name = f'{feat}_rbf'\n            df[rbf_name] = np.exp(-1.0 * df[feat]**2)  # Single gamma value\n            transformed_features.append(rbf_name)\n    \n    # 5. Limited ratio features\n    print(\"  - Creating limited ratio features...\")\n    ratio_count = 0\n    max_ratios = 15\n    for i, feat1 in enumerate(important_features[:10]):\n        if ratio_count >= max_ratios:\n            break\n        for feat2 in important_features[i+1:i+2]:  # Only next feature\n            if ratio_count >= max_ratios:\n                break\n            if feat1 in df.columns and feat2 in df.columns:\n                ratio_name = f'ratio_{feat1}_{feat2}'\n                df[ratio_name] = df[feat1] / (df[feat2] + 1e-10)\n                transformed_features.append(ratio_name)\n                ratio_count += 1\n    \n    # Clean up features\n    for col in transformed_features:\n        if col in df.columns:\n            df[col] = df[col].replace([np.inf, -np.inf], np.nan)\n            df[col] = df[col].fillna(0)\n    \n    print(f\"  Created {len(transformed_features)} complex transformations\")\n    return transformed_features[:max_features]\n\ndef create_memory_efficient_features(df, is_train=True):\n    \"\"\"Create features with memory efficiency in mind\"\"\"\n    features = []\n    \n    # Original features\n    base_features = [col for col in df.columns if col not in ['timestamp', 'label']]\n    features.extend(base_features)\n    \n    # Core market microstructure features only\n    if all(col in df.columns for col in ['bid_qty', 'ask_qty']):\n        df['bid_ask_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n        df['bid_ask_ratio'] = df['bid_qty'] / (df['ask_qty'] + 1e-10)\n        features.extend(['bid_ask_imbalance', 'bid_ask_ratio'])\n    \n    if all(col in df.columns for col in ['buy_qty', 'sell_qty']):\n        df['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n        df['net_order_flow'] = df['buy_qty'] - df['sell_qty']\n        features.extend(['order_flow_imbalance', 'net_order_flow'])\n    \n    # Limited rolling features - only most important columns and windows\n    key_cols = ['volume', 'bid_qty', 'ask_qty', 'net_order_flow']\n    key_cols = [col for col in key_cols if col in df.columns][:3]  # Maximum 3 columns\n    \n    windows = [10, 20, 50]  # Only 3 windows instead of 5\n    \n    for col in key_cols:\n        for window in windows:\n            # Only mean and std\n            df[f'{col}_roll_mean_{window}'] = df[col].shift(1).rolling(window, min_periods=1).mean()\n            df[f'{col}_roll_std_{window}'] = df[col].shift(1).rolling(window, min_periods=1).std()\n            df[f'{col}_z_score_{window}'] = (df[col] - df[f'{col}_roll_mean_{window}']) / (df[f'{col}_roll_std_{window}'] + 1e-10)\n            \n            features.extend([\n                f'{col}_roll_mean_{window}', \n                f'{col}_roll_std_{window}',\n                f'{col}_z_score_{window}'\n            ])\n    \n    # Add limited complex transformations\n    complex_features = create_selective_complex_transformations(df, base_features, max_features=Config.max_complex_features)\n    features.extend(complex_features)\n    \n    # Clean all features\n    print(f\"Total features created: {len(features)}\")\n    for col in features:\n        if col in df.columns:\n            df[col] = df[col].replace([np.inf, -np.inf], np.nan)\n            df[col] = df[col].fillna(0)\n    \n    return df, features\n\ndef select_features_multi_method(train_df, features, n_features=150):\n    \"\"\"Feature selection using time-weighted sample\"\"\"\n    print(f\"\\nSelecting top {n_features} features from {len(features)} using multiple methods...\")\n    \n    # Get time-weighted sample\n    sample_df = get_time_weighted_sample(train_df, Config.stratified_sample_size)\n    \n    X_sample = sample_df[features]\n    y_sample = sample_df['label']\n    \n    # Method 1: XGBoost feature importance\n    print(\"  Training XGBoost for feature importance...\")\n    xgb_model = XGBRegressor(\n        n_estimators=150,\n        max_depth=6,\n        learning_rate=0.1,\n        random_state=Config.random_seed,\n        n_jobs=-1\n    )\n    \n    xgb_model.fit(X_sample, y_sample)\n    xgb_importance = pd.DataFrame({\n        'feature': features,\n        'xgb_importance': xgb_model.feature_importances_\n    })\n    \n    # Method 2: Mutual information\n    print(\"  Calculating mutual information scores...\")\n    mi_scores = mutual_info_regression(X_sample, y_sample, random_state=Config.random_seed, n_neighbors=3)\n    mi_importance = pd.DataFrame({\n        'feature': features,\n        'mi_score': mi_scores\n    })\n    \n    # Method 3: Correlation (simplified)\n    print(\"  Calculating correlation scores...\")\n    correlations = []\n    for feat in features:\n        corr = abs(np.corrcoef(sample_df[feat].fillna(0), sample_df['label'])[0, 1])\n        correlations.append(corr)\n    \n    corr_importance = pd.DataFrame({\n        'feature': features,\n        'correlation': correlations\n    })\n    \n    # Combine rankings\n    importance_df = (xgb_importance\n                    .merge(mi_importance, on='feature')\n                    .merge(corr_importance, on='feature'))\n    \n    # Normalize and combine scores\n    for col in ['xgb_importance', 'mi_score', 'correlation']:\n        importance_df[f'{col}_rank'] = importance_df[col].rank(ascending=False)\n    \n    importance_df['combined_rank'] = (\n        importance_df['xgb_importance_rank'] * 0.5 +\n        importance_df['mi_score_rank'] * 0.3 +\n        importance_df['correlation_rank'] * 0.2\n    )\n    \n    importance_df = importance_df.sort_values('combined_rank')\n    top_features = importance_df.head(n_features)['feature'].tolist()\n    \n    print(f\"\\nTop 10 features by combined ranking:\")\n    for i, row in importance_df.head(10).iterrows():\n        print(f\"  {row['feature']}: XGB={row['xgb_importance']:.4f}, \"\n              f\"MI={row['mi_score']:.4f}, Corr={row['correlation']:.4f}\")\n    \n    return top_features\n\ndef get_memory_efficient_models():\n    \"\"\"Models optimized for memory efficiency while maintaining performance\"\"\"\n    models = {\n        # Efficient LightGBM\n        'lgb_efficient_v1': LGBMRegressor(\n            n_estimators=600,\n            num_leaves=60,\n            max_depth=10,\n            learning_rate=0.025,\n            feature_fraction=0.7,\n            bagging_fraction=0.7,\n            bagging_freq=5,\n            reg_alpha=3,\n            reg_lambda=3,\n            min_data_in_leaf=15,\n            random_state=Config.random_seed,\n            verbose=-1\n        ),\n        \n        'lgb_efficient_v2': LGBMRegressor(\n            n_estimators=500,\n            num_leaves=50,\n            max_depth=12,\n            learning_rate=0.03,\n            feature_fraction=0.75,\n            bagging_fraction=0.75,\n            bagging_freq=3,\n            reg_alpha=5,\n            reg_lambda=5,\n            min_data_in_leaf=20,\n            random_state=Config.random_seed + 1,\n            verbose=-1\n        ),\n        \n        # Efficient XGBoost\n        'xgb_efficient': XGBRegressor(\n            n_estimators=500,\n            max_depth=10,\n            learning_rate=0.025,\n            subsample=0.7,\n            colsample_bytree=0.7,\n            reg_alpha=5,\n            reg_lambda=5,\n            min_child_weight=5,\n            gamma=0.1,\n            random_state=Config.random_seed\n        ),\n        \n        # Efficient CatBoost\n        'cat_efficient': CatBoostRegressor(\n            iterations=500,\n            depth=10,\n            learning_rate=0.03,\n            l2_leaf_reg=8,\n            bagging_temperature=0.7,\n            random_strength=0.7,\n            border_count=128,\n            grow_policy='Lossguide',\n            random_state=Config.random_seed,\n            verbose=False\n        )\n    }\n    \n    return models\n\ndef train_with_validation_strategy(train_df, test_df, features):\n    \"\"\"Memory-efficient training with validation\"\"\"\n    X_train = train_df[features]\n    y_train = train_df['label']\n    X_test = test_df[features]\n    \n    # Remove NaN rows\n    valid_idx = ~(X_train.isna().any(axis=1) | y_train.isna())\n    X_train = X_train[valid_idx]\n    y_train = y_train[valid_idx]\n    \n    # Use StandardScaler for memory efficiency\n    scaler = StandardScaler()\n    X_train_scaled = pd.DataFrame(\n        scaler.fit_transform(X_train),\n        columns=X_train.columns,\n        index=X_train.index\n    )\n    X_test_scaled = pd.DataFrame(\n        scaler.transform(X_test),\n        columns=X_test.columns\n    )\n    \n    print(f\"\\nTraining on {len(X_train)} samples with {len(features)} features\")\n    \n    models = get_memory_efficient_models()\n    all_predictions = {}\n    model_scores = {}\n    \n    # Use KFold validation\n    kf = KFold(n_splits=Config.n_folds, shuffle=True, random_state=Config.random_seed)\n    \n    for model_name, model in models.items():\n        print(f\"\\nTraining {model_name}...\")\n        fold_predictions = []\n        fold_scores = []\n        \n        for fold, (train_idx, val_idx) in enumerate(kf.split(X_train_scaled)):\n            X_fold_train = X_train_scaled.iloc[train_idx]\n            y_fold_train = y_train.iloc[train_idx]\n            X_fold_val = X_train_scaled.iloc[val_idx]\n            y_fold_val = y_train.iloc[val_idx]\n            \n            # Time-based sample weights\n            sample_weights = np.linspace(0.8, 1.2, len(X_fold_train))\n            \n            # Clone and train model\n            model_clone = model.__class__(**model.get_params())\n            \n            if 'xgb' in model_name:\n                model_clone.fit(\n                    X_fold_train, y_fold_train,\n                    sample_weight=sample_weights,\n                    eval_set=[(X_fold_val, y_fold_val)],\n                    early_stopping_rounds=50,\n                    verbose=False\n                )\n            elif 'lgb' in model_name:\n                model_clone.fit(\n                    X_fold_train, y_fold_train,\n                    sample_weight=sample_weights,\n                    eval_set=[(X_fold_val, y_fold_val)],\n                    callbacks=[lgbm.early_stopping(50, verbose=False)],\n                    eval_metric='rmse'\n                )\n            else:\n                model_clone.fit(X_fold_train, y_fold_train, sample_weight=sample_weights)\n            \n            # Validate\n            val_pred = model_clone.predict(X_fold_val)\n            val_score = pearsonr(y_fold_val, val_pred)[0]\n            fold_scores.append(val_score)\n            \n            # Predict on test\n            test_pred = model_clone.predict(X_test_scaled)\n            fold_predictions.append(test_pred)\n            \n            print(f\"  Fold {fold + 1}: {val_score:.4f}\")\n            \n            # Clear some memory\n            del X_fold_train, X_fold_val, y_fold_train, y_fold_val\n        \n        # Average predictions across folds\n        model_predictions = np.mean(fold_predictions, axis=0)\n        model_score = np.mean(fold_scores)\n        \n        all_predictions[model_name] = model_predictions\n        model_scores[model_name] = model_score\n        \n        print(f\"  Average score: {model_score:.4f}\")\n    \n    return all_predictions, model_scores\n\ndef create_ensemble(predictions_dict, scores_dict):\n    \"\"\"Create memory-efficient ensemble\"\"\"\n    print(f\"\\nCreating ensemble from {len(predictions_dict)} models\")\n    \n    # Simple weighted average based on scores\n    scores = np.array(list(scores_dict.values()))\n    weights = scores ** 2  # Square for emphasis\n    weights = weights / weights.sum()\n    \n    predictions_array = np.array(list(predictions_dict.values()))\n    weighted_ensemble = np.average(predictions_array, axis=0, weights=weights)\n    \n    # Top 2 models average\n    top_2 = sorted(scores_dict.items(), key=lambda x: x[1], reverse=True)[:2]\n    top_predictions = np.array([predictions_dict[k] for k, _ in top_2])\n    top_ensemble = np.mean(top_predictions, axis=0)\n    \n    return weighted_ensemble, top_ensemble\n\ndef main():\n    \"\"\"Main execution pipeline optimized for memory efficiency\"\"\"\n    print(\"=\"*60)\n    print(\"MEMORY-OPTIMIZED CRYPTO PREDICTION PIPELINE\")\n    print(\"=\"*60)\n    \n    # Load data\n    train, test = load_and_prepare_data()\n    \n    # Create memory-efficient features\n    print(\"\\nCreating memory-efficient features...\")\n    train, train_features = create_memory_efficient_features(train, is_train=True)\n    test, test_features = create_memory_efficient_features(test, is_train=False)\n    \n    # Ensure same features\n    common_features = list(set(train_features) & set(test_features))\n    print(f\"Common features: {len(common_features)}\")\n    \n    # Remove highly correlated features\n    common_features = remove_correlated_features(train, common_features, Config.correlation_threshold)\n    \n    # Select top features\n    top_features = select_features_multi_method(train, common_features, n_features=Config.n_top_features)\n    \n    # Train models\n    predictions, scores = train_with_validation_strategy(train, test, top_features)\n    \n    # Create ensemble\n    print(\"\\n\" + \"=\"*60)\n    print(\"CREATING ENSEMBLE PREDICTIONS\")\n    print(\"=\"*60)\n    \n    weighted_ensemble, top_ensemble = create_ensemble(predictions, scores)\n    \n    # Final ensemble\n    final_ensemble = 0.7 * weighted_ensemble + 0.3 * top_ensemble\n    \n    # Save submissions\n    sample = pd.read_csv(Config.sample_sub_path)\n    \n    print(\"\\nModel Performance Summary:\")\n    print(\"-\" * 50)\n    for model, score in sorted(scores.items(), key=lambda x: x[1], reverse=True):\n        print(f\"{model}: {score:.4f}\")\n    print(\"-\" * 50)\n    \n    # Save best single model\n    best_model = max(scores.items(), key=lambda x: x[1])[0]\n    submission = sample.copy()\n    submission['prediction'] = predictions[best_model]\n    submission.to_csv('submission_best_memory_opt.csv', index=False)\n    print(f\"\\nSaved: submission_best_memory_opt.csv ({best_model})\")\n    \n    # Save weighted ensemble\n    submission = sample.copy()\n    submission['prediction'] = weighted_ensemble\n    submission.to_csv('submission_weighted_memory_opt.csv', index=False)\n    print(\"Saved: submission_weighted_memory_opt.csv\")\n    \n    # Save final ensemble\n    submission = sample.copy()\n    submission['prediction'] = final_ensemble\n    submission.to_csv('submission_final_memory_opt.csv', index=False)\n    print(\"Saved: submission_final_memory_opt.csv (RECOMMENDED)\")\n    \n    print(\"\\n\" + \"=\"*60)\n    print(\"MEMORY-OPTIMIZED PIPELINE COMPLETE!\")\n    print(f\"Successfully processed with reduced memory footprint\")\n    print(\"=\"*60)\n\n# Import for LightGBM\nimport lightgbm as lgbm\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}