{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-13T07:33:46.977365Z","iopub.execute_input":"2025-06-13T07:33:46.978204Z","iopub.status.idle":"2025-06-13T07:33:47.432702Z","shell.execute_reply.started":"2025-06-13T07:33:46.978161Z","shell.execute_reply":"2025-06-13T07:33:47.431573Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Import Libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport gc # For garbage collection\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom xgboost import XGBRegressor\nfrom scipy.stats import pearsonr\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T07:33:47.434464Z","iopub.execute_input":"2025-06-13T07:33:47.434984Z","iopub.status.idle":"2025-06-13T07:33:48.406378Z","shell.execute_reply.started":"2025-06-13T07:33:47.434957Z","shell.execute_reply":"2025-06-13T07:33:48.405437Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1. Setup and Data Loading","metadata":{}},{"cell_type":"code","source":"# Define file paths (assuming you are in a Kaggle notebook environment)\nTRAIN_PATH = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\nTEST_PATH = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\nSAMPLE_SUB_PATH = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T07:33:48.407439Z","iopub.execute_input":"2025-06-13T07:33:48.407841Z","iopub.status.idle":"2025-06-13T07:33:48.413304Z","shell.execute_reply.started":"2025-06-13T07:33:48.407816Z","shell.execute_reply":"2025-06-13T07:33:48.412241Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load data\nprint(\"Loading training data...\")\ntrain_df = pd.read_parquet(TRAIN_PATH, engine='pyarrow')\nprint(\"Loading test data...\")\ntest_df = pd.read_parquet(TEST_PATH, engine='pyarrow')\nprint(\"Loading sample submission...\")\nsample_submission = pd.read_csv(SAMPLE_SUB_PATH)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T07:33:48.415619Z","iopub.execute_input":"2025-06-13T07:33:48.415896Z","iopub.status.idle":"2025-06-13T07:34:30.943814Z","shell.execute_reply.started":"2025-06-13T07:33:48.415874Z","shell.execute_reply":"2025-06-13T07:34:30.942757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Train data shape: {train_df.shape}\")\nprint(f\"Test data shape: {test_df.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T07:34:30.944927Z","iopub.execute_input":"2025-06-13T07:34:30.945359Z","iopub.status.idle":"2025-06-13T07:34:30.951076Z","shell.execute_reply.started":"2025-06-13T07:34:30.945305Z","shell.execute_reply":"2025-06-13T07:34:30.949912Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Memory Reduction (Crucial for Large Datasets)","metadata":{}},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    start_mem = df.memory_usage().sum() / 1024**2\n    print(f'Memory usage before: {start_mem:.2f} MB')\n    for col in df.columns:\n        col_type = df[col].dtype\n        if 'float' in str(col_type):\n            if df[col].min() > np.finfo(np.float16).min and df[col].max() < np.finfo(np.float16).max:\n                df[col] = df[col].astype(np.float16)\n            elif df[col].min() > np.finfo(np.float32).min and df[col].max() < np.finfo(np.float32).max:\n                df[col] = df[col].astype(np.float32)\n        elif 'int' in str(col_type):\n            if df[col].min() > np.iinfo(np.int8).min and df[col].max() < np.iinfo(np.int8).max:\n                df[col] = df[col].astype(np.int8)\n            elif df[col].min() > np.iinfo(np.int16).min and df[col].max() < np.iinfo(np.int16).max:\n                df[col] = df[col].astype(np.int16)\n            elif df[col].min() > np.iinfo(np.int32).min and df[col].max() < np.iinfo(np.int32).max:\n                df[col] = df[col].astype(np.int32)\n            elif df[col].min() > np.iinfo(np.int64).min and df[col].max() < np.iinfo(np.int64).max:\n                df[col] = df[col].astype(np.int64)\n    end_mem = df.memory_usage().sum() / 1024**2\n    print(f'Memory usage after: {end_mem:.2f} MB ({100 * (start_mem - end_mem) / start_mem:.1f}% reduction)')\n    return df\n\ntrain_df = reduce_mem_usage(train_df)\ntest_df = reduce_mem_usage(test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T07:34:30.952112Z","iopub.execute_input":"2025-06-13T07:34:30.952404Z","iopub.status.idle":"2025-06-13T07:34:47.299071Z","shell.execute_reply.started":"2025-06-13T07:34:30.952376Z","shell.execute_reply":"2025-06-13T07:34:47.298148Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Preprocessing and Feature Engineering","metadata":{}},{"cell_type":"code","source":"# Replace inf/-inf with NaN, then fill NaNs (e.g., with 0 or median)\nfor df in [train_df, test_df]:\n    df.replace([np.inf, -np.inf], np.nan, inplace=True)\n    df.fillna(0, inplace=True) # A common strategy for financial data, but consider alternatives like median or forward/backward fill\n\n# Drop columns with only one unique value in the training set\nnunique_cols = train_df.nunique()\ncols_to_drop = nunique_cols[nunique_cols == 1].index.tolist()\nprint(f\"Dropping {len(cols_to_drop)} columns with only one unique value: {cols_to_drop}\")\ntrain_df.drop(columns=cols_to_drop, inplace=True)\n\n# Ensure the same columns are dropped from the test set (excluding 'label' if it was in cols_to_drop from train)\ntest_cols_to_drop = [col for col in cols_to_drop if col != 'label'] # 'label' won't be in test features\ntest_df.drop(columns=test_cols_to_drop, inplace=True)\n\n\n# Feature Engineering Examples (Highly customizable and dataset-specific)\n# These are just examples, you'll need to experiment a lot!\n\ndef create_features(df):\n    # Basic interactions\n    df['bid_ask_spread'] = df['ask_qty'] - df['bid_qty']\n    df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-6) # Add small epsilon to avoid division by zero\n    df['total_qty'] = df['bid_qty'] + df['ask_qty'] + df['buy_qty'] + df['sell_qty']\n\n    # Example: Simple rolling features on 'volume'\n    # For actual implementation, consider the 'timestamp' and proper time-series splits\n    # and avoiding future leakage. This is a simplified example.\n    df['volume_rolling_mean_5'] = df['volume'].rolling(window=5, min_periods=1).mean()\n    df['volume_rolling_std_5'] = df['volume'].rolling(window=5, min_periods=1).std()\n\n    # You could also consider more complex features like:\n    # - Lagged values of 'label' (if you were doing multi-step forecasting, but here it's predicting future label)\n    # - Statistical features (min, max, median, skew, kurtosis) over rolling windows for various X features.\n    # - Fourier Transforms or Wavelet Transforms for extracting cyclical patterns.\n    # - Technical indicators (if you map X features to known financial concepts).\n\n    return df\n\n# Apply feature engineering\n# Be careful with applying rolling features across the entire dataset if you are doing time-series cross-validation\n# For the sake of this example, we apply it directly.\n# In a real competition, you'd apply these during the split for proper CV.\nprint(\"Creating features for training data...\")\ntrain_df = create_features(train_df)\nprint(\"Creating features for test data...\")\ntest_df = create_features(test_df)\n\n# Drop original timestamp from features as it's not a direct numerical feature for many models\n# Keep it as index if you are doing time-series specific operations\nif 'timestamp' in train_df.columns:\n    train_df.set_index('timestamp', inplace=True)\nif 'timestamp' in test_df.columns:\n    test_df.set_index('timestamp', inplace=True)\n\nprint(f\"Train data shape after feature engineering: {train_df.shape}\")\nprint(f\"Test data shape after feature engineering: {test_df.shape}\")\n\n# Align columns - crucial if feature engineering creates different columns or if some columns were dropped\ntrain_labels = train_df['label']\ntrain_features = train_df.drop(columns=['label'])\ntest_features = test_df.drop(columns=['label'], errors='ignore') # 'label' might not exist in test_df\n\n# Get common columns after feature engineering and dropping\ncommon_cols = list(set(train_features.columns) & set(test_features.columns))\ntrain_features = train_features[common_cols]\ntest_features = test_features[common_cols]\n\nprint(f\"Train features shape after aligning: {train_features.shape}\")\nprint(f\"Test features shape after aligning: {test_features.shape}\")\n\n# Clean up memory\ndel train_df, test_df\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T07:34:47.299876Z","iopub.execute_input":"2025-06-13T07:34:47.300108Z","iopub.status.idle":"2025-06-13T07:35:24.367301Z","shell.execute_reply.started":"2025-06-13T07:34:47.300090Z","shell.execute_reply":"2025-06-13T07:35:24.366267Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Model Training","metadata":{}},{"cell_type":"code","source":"# Define features (X) and target (y)\nX = train_features\ny = train_labels\n\n# Split data into training and validation sets (using a simple split for quick demonstration)\n# For time series, a sequential split is often preferred:\n# X_train, X_val, y_train, y_val = X.iloc[:-val_size], X.iloc[-val_size:], y.iloc[:-val_size], y.iloc[-val_size:]\n# For this example, we'll use a random split for simplicity.\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42, shuffle=False) # Keep shuffle=False for time series\nprint(f\"X_train shape: {X_train.shape}, y_train shape: {y_train.shape}\")\nprint(f\"X_val shape: {X_val.shape}, y_val shape: {y_val.shape}\")\n\n# Initialize and train an XGBoost Regressor\n# Parameters can be tuned extensively\nmodel = XGBRegressor(\n    objective='reg:squarederror',\n    n_estimators=500,\n    learning_rate=0.05,\n    max_depth=7,\n    subsample=0.7,\n    colsample_bytree=0.7,\n    random_state=42,\n    n_jobs=-1, # Use all available CPU cores\n    tree_method='hist', # Faster for large datasets\n    # enable_categorical=True # If you had categorical features and wanted to use XGBoost's native handling\n)\n\nprint(\"Training model...\")\nmodel.fit(X_train, y_train,\n          eval_set=[(X_val, y_val)],\n          early_stopping_rounds=50, # Stop if validation error doesn't improve for 50 rounds\n          verbose=False) # Set to True for verbose output during training\n\nprint(\"Model training complete.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T07:35:24.368555Z","iopub.execute_input":"2025-06-13T07:35:24.368932Z","iopub.status.idle":"2025-06-13T07:38:23.148475Z","shell.execute_reply.started":"2025-06-13T07:35:24.368905Z","shell.execute_reply":"2025-06-13T07:38:23.147411Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. Prediction and Submission","metadata":{}},{"cell_type":"code","source":"# Predict on the validation set\nval_preds = model.predict(X_val)\nval_corr, _ = pearsonr(y_val, val_preds)\nprint(f\"Validation Pearson Correlation Coefficient: {val_corr:.4f}\")\n\n# Predict on the actual test set\nprint(\"Generating predictions on test data...\")\ntest_predictions = model.predict(test_features)\n\n# Create submission file\nsample_submission['prediction'] = test_predictions\nsubmission_file_name = 'submission.csv'\nsample_submission.to_csv(submission_file_name, index=False)\n\nprint(f\"Submission file '{submission_file_name}' created successfully.\")\nprint(sample_submission.head())\n\n# Optional: Visualize a small sample of predictions vs actuals (from validation set)\nplt.figure(figsize=(12, 6))\nplt.plot(y_val.values[:200], label='Actual Label')\nplt.plot(val_preds[:200], label='Predicted Label')\nplt.title('Validation Predictions vs Actuals (First 200 points)')\nplt.xlabel('Time Step')\nplt.ylabel('Label Value')\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T07:38:23.149596Z","iopub.execute_input":"2025-06-13T07:38:23.149942Z","iopub.status.idle":"2025-06-13T07:38:31.664714Z","shell.execute_reply.started":"2025-06-13T07:38:23.149907Z","shell.execute_reply":"2025-06-13T07:38:31.663406Z"}},"outputs":[],"execution_count":null}]}