{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Suppress warnings\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n# Memory management\nimport gc\nimport psutil\nimport os\n\n# Data manipulation\nimport pandas as pd\nimport numpy as np\n\n# Modeling\nimport xgboost as xgb\n\n# Evaluation\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score\nfrom sklearn.preprocessing import StandardScaler\n\n# Visualization\nimport matplotlib.pyplot as plt\n\n# Display\nfrom IPython.display import display\nimport time\n\ndef print_memory_usage():\n    \"\"\"Print current memory usage\"\"\"\n    process = psutil.Process(os.getpid())\n    memory_mb = process.memory_info().rss / 1024 / 1024\n    print(f\"💾 Current memory usage: {memory_mb:.2f} MB\")\n\ndef check_gpu_availability():\n    \"\"\"Check for GPU availability and return device type\"\"\"\n    try:\n        # Check if XGBoost was compiled with GPU support\n        import xgboost as xgb\n        # Try to create a GPU-enabled booster\n        dtrain = xgb.DMatrix(np.random.rand(10, 5), label=np.random.rand(10))\n        params = {'tree_method': 'gpu_hist', 'gpu_id': 0}\n        bst = xgb.train(params, dtrain, num_boost_round=1, verbose_eval=False)\n        print(\"🚀 GPU detected and will be used for training\")\n        return 'gpu'\n    except Exception as e:\n        print(\"⚠️  GPU not available, using CPU with multi-threading\")\n        print(f\"   GPU error: {str(e)}\")\n        return 'cpu'\n\ndef get_gpu_count():\n    \"\"\"Get number of available GPUs\"\"\"\n    try:\n        import subprocess\n        result = subprocess.run(['nvidia-smi', '--list-gpus'], \n                              capture_output=True, text=True, timeout=5)\n        if result.returncode == 0:\n            gpu_count = len(result.stdout.strip().split('\\n'))\n            print(f\"🎮 Found {gpu_count} GPU(s)\")\n            return gpu_count\n        else:\n            return 0\n    except:\n        return 0\n\nprint(\"🔄 Starting DRW Crypto Market Prediction with XGBoost\")\nprint(\"=\" * 60)\n\n# Check system resources\nprint(\"🔍 Checking system resources...\")\ndevice_type = check_gpu_availability()\ncpu_count = os.cpu_count()\ngpu_count = get_gpu_count() if device_type == 'gpu' else 0\nprint(f\"💻 Available CPU cores: {cpu_count}\")\nif gpu_count > 0:\n    print(f\"🎮 Available GPUs: {gpu_count}\")\nprint_memory_usage()\nprint()\n\nprint(\"📊 Loading training data...\")\nstart_time = time.time()\ntrain = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\nprint(f\"✅ Training data loaded in {time.time() - start_time:.2f} seconds\")\nprint(f\"📈 Training data shape: {train.shape}\")\nprint_memory_usage()\nprint()\n\ndef preprocess_train(df):\n    \"\"\"Preprocess training data with feature engineering\"\"\"\n    print(\"🔧 Starting feature engineering on training data...\")\n    \n    # Basic features\n    print(\"   - Creating imbalance features...\")\n    df['imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-6)\n    \n    print(\"   - Creating spread features...\")\n    df['bid_ask_spread'] = df['ask_qty'] - df['bid_qty']\n    df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-6)\n    \n    # Additional features for better performance\n    print(\"   - Creating volume-based features...\")\n    df['volume_imbalance'] = df['volume'] * df['imbalance']\n    df['price_pressure'] = (df['buy_qty'] - df['sell_qty']) / df['volume']\n    \n    print(\"   - Creating momentum features...\")\n    df['bid_ask_mid'] = (df['bid_qty'] + df['ask_qty']) / 2\n    df['order_flow'] = df['buy_qty'] - df['sell_qty']\n    \n    # XGBoost-specific features (interaction terms)\n    print(\"   - Creating interaction features...\")\n    df['volume_price_pressure'] = df['volume'] * df['price_pressure']\n    df['imbalance_squared'] = df['imbalance'] ** 2\n    \n    print(\"   - Handling infinite values and NaNs...\")\n    df.replace([np.inf, -np.inf], np.nan, inplace=True)\n    df.fillna(0, inplace=True)\n\n    # Define feature columns\n    base_features = [f'X{i}' for i in range(1, 891)]\n    market_features = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']\n    engineered_features = [\n        'imbalance', 'bid_ask_spread', 'buy_sell_ratio', 'volume_imbalance',\n        'price_pressure', 'bid_ask_mid', 'order_flow', 'volume_price_pressure',\n        'imbalance_squared'\n    ]\n    \n    features = base_features + market_features + engineered_features\n    \n    print(\"   - Scaling features...\")\n    scaler = StandardScaler()\n    X = scaler.fit_transform(df[features].astype(np.float32))\n    y = df['label'].astype(np.float32).values\n\n    print(f\"✅ Feature engineering completed. Final feature count: {len(features)}\")\n    return X, y, features, scaler\n\n# Preprocess training data\nX, y, features, scaler = preprocess_train(train)\nprint_memory_usage()\n\n# Delete training dataframe to free memory\nprint(\"🗑️  Deleting training dataframe to free memory...\")\ndel train\ngc.collect()\nprint_memory_usage()\nprint()\n\nprint(\"🤖 Setting up XGBoost model...\")\n\n# XGBoost parameters optimized for financial data and GPU usage\nif device_type == 'gpu':\n    xgb_params = {\n        'objective': 'reg:squarederror',\n        'eval_metric': 'rmse',\n        'tree_method': 'gpu_hist',\n        'gpu_id': 0,\n        'predictor': 'gpu_predictor',\n        'max_depth': 8,\n        'learning_rate': 0.1,\n        'subsample': 0.8,\n        'colsample_bytree': 0.8,\n        'colsample_bylevel': 0.8,\n        'min_child_weight': 3,\n        'reg_alpha': 0.1,\n        'reg_lambda': 1.0,\n        'random_state': 42,\n        'verbosity': 1,\n        'n_jobs': 1,  # GPU handles parallelization\n    }\nelse:\n    xgb_params = {\n        'objective': 'reg:squarederror',\n        'eval_metric': 'rmse',\n        'tree_method': 'hist',\n        'max_depth': 8,\n        'learning_rate': 0.1,\n        'subsample': 0.8,\n        'colsample_bytree': 0.8,\n        'colsample_bylevel': 0.8,\n        'min_child_weight': 3,\n        'reg_alpha': 0.1,\n        'reg_lambda': 1.0,\n        'random_state': 42,\n        'verbosity': 1,\n        'n_jobs': cpu_count,\n    }\n\nprint(f\"📋 XGBoost parameters: {xgb_params}\")\n\n# Define correlation evaluation metric for XGBoost\ndef xgb_correlation_eval(y_pred, dtrain):\n    \"\"\"Custom correlation evaluation metric for XGBoost\"\"\"\n    y_true = dtrain.get_label()\n    correlation = np.corrcoef(y_pred, y_true)[0, 1]\n    # Return name and correlation (higher is better, so we don't negate)\n    return 'correlation', correlation\n\n# Create DMatrix for XGBoost (memory efficient)\nprint(\"🔄 Creating XGBoost DMatrix...\")\ndtrain = xgb.DMatrix(X, label=y, feature_names=[f'feature_{i}' for i in range(X.shape[1])])\nprint_memory_usage()\n\n# Delete X, y to free memory before training\nprint(\"🗑️  Deleting training arrays to free memory...\")\ndel X, y\ngc.collect()\nprint_memory_usage()\nprint()\n\n# Create and train model\nprint(\"🏋️  Training XGBoost model...\")\nstart_time = time.time()\n\n# Update params to use correlation\nxgb_params['eval_metric'] = ['rmse']  # Keep RMSE for monitoring, add correlation via feval\n\nmodel = xgb.train(\n    params=xgb_params,\n    dtrain=dtrain,\n    num_boost_round=1000,\n    evals=[(dtrain, 'train')],\n    feval=xgb_correlation_eval,\n    early_stopping_rounds=50,\n    verbose_eval=100,\n    maximize=True  # Since correlation should be maximized\n)\n\ntraining_time = time.time() - start_time\nprint(f\"✅ Model training completed in {training_time:.2f} seconds\")\nprint(f\"🎯 Best iteration: {model.best_iteration}\")\nprint(f\"🎯 Best score: {model.best_score}\")\nprint_memory_usage()\n\n# Delete training DMatrix to free memory\nprint(\"🗑️  Deleting training DMatrix to free memory...\")\ndel dtrain\ngc.collect()\nprint_memory_usage()\nprint()\n\nprint(\"📊 Loading test data...\")\nstart_time = time.time()\ntest = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\nprint(f\"✅ Test data loaded in {time.time() - start_time:.2f} seconds\")\nprint(f\"📈 Test data shape: {test.shape}\")\nprint_memory_usage()\n\ndef preprocess_test(df_test, features, scaler):\n    \"\"\"Preprocess test data with same feature engineering as training\"\"\"\n    print(\"🔧 Starting feature engineering on test data...\")\n    \n    # Apply same feature engineering as training\n    print(\"   - Creating imbalance features...\")\n    df_test['imbalance'] = (df_test['buy_qty'] - df_test['sell_qty']) / (df_test['buy_qty'] + df_test['sell_qty'] + 1e-6)\n    \n    print(\"   - Creating spread features...\")\n    df_test['bid_ask_spread'] = df_test['ask_qty'] - df_test['bid_qty']\n    df_test['buy_sell_ratio'] = df_test['buy_qty'] / (df_test['sell_qty'] + 1e-6)\n    \n    print(\"   - Creating volume-based features...\")\n    df_test['volume_imbalance'] = df_test['volume'] * df_test['imbalance']\n    df_test['price_pressure'] = (df_test['buy_qty'] - df_test['sell_qty']) / df_test['volume']\n    \n    print(\"   - Creating momentum features...\")\n    df_test['bid_ask_mid'] = (df_test['bid_qty'] + df_test['ask_qty']) / 2\n    df_test['order_flow'] = df_test['buy_qty'] - df_test['sell_qty']\n    \n    # XGBoost-specific features (interaction terms)\n    print(\"   - Creating interaction features...\")\n    df_test['volume_price_pressure'] = df_test['volume'] * df_test['price_pressure']\n    df_test['imbalance_squared'] = df_test['imbalance'] ** 2\n    \n    print(\"   - Handling infinite values and NaNs...\")\n    df_test.replace([np.inf, -np.inf], np.nan, inplace=True)\n    df_test.fillna(0, inplace=True)\n\n    print(\"   - Scaling features...\")\n    X_test = scaler.transform(df_test[features].astype(np.float32))\n    \n    print(\"✅ Test data preprocessing completed\")\n    return X_test\n\n# Preprocess test data\nX_test = preprocess_test(test, features, scaler)\nprint_memory_usage()\n\n# Delete test dataframe to free memory (keep only what we need for submission)\nprint(\"🗑️  Cleaning up test data to free memory...\")\ntest_ids = test.index if 'id' not in test.columns else test['id']\ndel test\ngc.collect()\nprint_memory_usage()\nprint()\n\nprint(\"🔮 Making predictions on test data...\")\nstart_time = time.time()\n\n# Create DMatrix for test data\ndtest = xgb.DMatrix(X_test, feature_names=[f'feature_{i}' for i in range(X_test.shape[1])])\n\n# Make predictions\ntest_preds = model.predict(dtest, iteration_range=(0, model.best_iteration))\n\nprediction_time = time.time() - start_time\nprint(f\"✅ Predictions completed in {prediction_time:.2f} seconds\")\nprint(f\"📊 Prediction statistics:\")\nprint(f\"   - Min: {test_preds.min():.6f}\")\nprint(f\"   - Max: {test_preds.max():.6f}\")\nprint(f\"   - Mean: {test_preds.mean():.6f}\")\nprint(f\"   - Std: {test_preds.std():.6f}\")\n\n# Delete test features and model to free memory\nprint(\"🗑️  Deleting test features and model to free memory...\")\ndel X_test, dtest, model, scaler\ngc.collect()\nprint_memory_usage()\nprint()\n\nprint(\"📝 Creating submission file...\")\n# Load sample submission to get the correct format\nsubmission = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\nprint(f\"📋 Submission template shape: {submission.shape}\")\n\n# Update predictions\nsubmission['prediction'] = test_preds\n\n# Save submission\nsubmission.to_csv(\"submission.csv\", index=False)\nprint(\"✅ Submission file saved as submission.csv\")\n\n# Final cleanup\ndel submission, test_preds\ngc.collect()\n\nprint()\nprint(\"🎉 Process completed successfully!\")\nprint(\"=\" * 60)\nprint_memory_usage()\nprint(f\"📁 Submission file: submission.csv\")\nprint(\"🚀 Ready for submission!\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null}]}