{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-02T02:59:25.478112Z","iopub.execute_input":"2025-07-02T02:59:25.479061Z","iopub.status.idle":"2025-07-02T02:59:25.485172Z","shell.execute_reply.started":"2025-07-02T02:59:25.479026Z","shell.execute_reply":"2025-07-02T02:59:25.484382Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1: Beginner - LightGBM on CPU with Basic Memory Optimization","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport lightgbm as lgb\nfrom sklearn.model_selection import train_test_split\nfrom scipy.stats import pearsonr\nimport gc\nimport warnings\n\nwarnings.filterwarnings('ignore')\n\n# 1. Load Data\n# Assuming train.parquet and test.parquet are in '../input/drw-crypto-market-prediction/'\nprint(\"Loading data...\")\ntrain_df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\ntest_df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\nsample_submission = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\nprint(\"Data loaded.\")\n\n# 2. Basic Memory Reduction Function\ndef reduce_mem_usage(df):\n    start_mem = df.memory_usage(deep=True).sum() / 1024**2\n    print(f\"Memory usage of dataframe before reduction: {start_mem:.2f} MB\")\n    for col in df.columns:\n        col_type = df[col].dtype\n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else: # floats\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n    end_mem = df.memory_usage(deep=True).sum() / 1024**2\n    print(f\"Memory usage of dataframe after reduction: {end_mem:.2f} MB\")\n    return df\n\ntrain_df = reduce_mem_usage(train_df)\ntest_df = reduce_mem_usage(test_df)\n\n# 3. Feature Engineering (Basic)\ndef engineer_features(df):\n    df['bid_ask_spread'] = df['ask_qty'] - df['bid_qty']\n    df['total_qty'] = df['bid_qty'] + df['ask_qty']\n    df['buy_sell_diff'] = df['buy_qty'] - df['sell_qty']\n    df['volume_per_total_qty'] = df['volume'] / (df['total_qty'] + 1e-6) # Add epsilon to avoid division by zero\n    return df\n\nprint(\"Engineering features...\")\ntrain_df = engineer_features(train_df)\ntest_df = engineer_features(test_df)\nprint(\"Features engineered.\")\n\n# 4. Define Features and Target\nTARGET = 'label'\nfeatures = [col for col in train_df.columns if col not in ['ID', 'timestamp', TARGET]]\n\n# Drop columns with only one unique value (these provide no predictive power)\n# and handle 'label' column in test_df (it's not present during actual test inference)\nnunique_cols = [col for col in train_df.columns if train_df[col].nunique() == 1]\ntrain_df.drop(columns=nunique_cols, inplace=True)\nfeatures = [f for f in features if f not in nunique_cols]\n# Ensure test_df also has these dropped\nif TARGET in test_df.columns:\n    test_df.drop(columns=[TARGET] + nunique_cols, inplace=True)\nelse:\n    test_df.drop(columns=nunique_cols, inplace=True)\n\n# Align columns - very important for robust predictions\ntrain_cols = set(train_df.columns)\ntest_cols = set(test_df.columns)\ncommon_features = list(train_cols.intersection(test_cols) - {TARGET, 'ID', 'timestamp'})\nfeatures = [f for f in features if f in common_features] # Only use features present in both train and test\n\nX = train_df[features]\ny = train_df[TARGET]\n\ndel train_df # Free up memory\ngc.collect()\n\n# 5. Time Series Split (Crucial for financial data)\n# Use a simple time-based split for initial exploration\n# The data is already sorted by timestamp.\nsplit_index = int(len(X) * 0.8)\nX_train, X_val = X.iloc[:split_index], X.iloc[split_index:]\ny_train, y_val = y.iloc[:split_index], y.iloc[split_index:]\n\ndel X, y # Free up memory\ngc.collect()\n\nprint(f\"Train data shape: {X_train.shape}\")\nprint(f\"Validation data shape: {X_val.shape}\")\n\n# 6. LightGBM Model Training (CPU)\nprint(\"Training LightGBM model (CPU)...\")\nlgb_params = {\n    'objective': 'regression_l1', # MAE is often robust for financial predictions\n    'metric': 'rmse', # RMSE for internal evaluation\n    'n_estimators': 1000,\n    'learning_rate': 0.05,\n    'feature_fraction': 0.8,\n    'bagging_fraction': 0.8,\n    'bagging_freq': 1,\n    'lambda_l1': 0.1,\n    'lambda_l2': 0.1,\n    'num_leaves': 31,\n    'verbose': -1,\n    'n_jobs': -1, # Use all available CPU cores\n    'seed': 42,\n    'boosting_type': 'gbdt',\n}\n\nmodel = lgb.LGBMRegressor(**lgb_params)\n\nmodel.fit(X_train, y_train,\n          eval_set=[(X_val, y_val)],\n          eval_metric='rmse', # Use RMSE for early stopping\n          callbacks=[lgb.early_stopping(100, verbose=False)], # Early stopping rounds\n          )\nprint(\"LightGBM training complete.\")\n\n# 7. Prediction and Evaluation\nval_preds = model.predict(X_val)\ncorrelation, _ = pearsonr(val_preds, y_val)\nprint(f\"Validation Pearson Correlation: {correlation:.4f}\")\n\n# 8. Generate Test Predictions\nprint(\"Generating test predictions...\")\ntest_preds = model.predict(test_df[features])\n\n# Clip predictions to the expected range if specified in competition\n# (e.g., between -5.0 and 5.0 as seen in some discussions)\ntest_preds = np.clip(test_preds, -5.0, 5.0)\n\n# Create submission file\nsubmission_df = pd.DataFrame({'ID': sample_submission['ID'], 'prediction': test_preds})\nsubmission_df.to_csv('submission1.csv', index=False)\nprint(\"Submission file created: submission1.csv\")\n\ndel X_train, X_val, y_train, y_val, test_df, model, test_preds # Clean up memory\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T02:59:26.038265Z","iopub.execute_input":"2025-07-02T02:59:26.039070Z","iopub.status.idle":"2025-07-02T03:01:52.178038Z","shell.execute_reply.started":"2025-07-02T02:59:26.039041Z","shell.execute_reply":"2025-07-02T03:01:52.177374Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2: Intermediate - LightGBM on GPU with Enhanced Memory Management","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport lightgbm as lgb\nfrom sklearn.model_selection import train_test_split\nfrom scipy.stats import pearsonr\nimport gc\nimport warnings\nimport torch # Just to check for CUDA availability if needed for other models\n\nwarnings.filterwarnings('ignore')\n\n# 1. Load Data and Memory Reduction (same as Level 1)\nprint(\"Loading data...\")\ntrain_df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\ntest_df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\nsample_submission = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\nprint(\"Data loaded.\")\n\ndef reduce_mem_usage(df):\n    start_mem = df.memory_usage(deep=True).sum() / 1024**2\n    print(f\"Memory usage of dataframe before reduction: {start_mem:.2f} MB\")\n    for col in df.columns:\n        col_type = df[col].dtype\n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max: # Use float16 for floats\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else: # floats\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n    end_mem = df.memory_usage(deep=True).sum() / 1024**2\n    print(f\"Memory usage of dataframe after reduction: {end_mem:.2f} MB\")\n    return df\n\ntrain_df = reduce_mem_usage(train_df)\ntest_df = reduce_mem_usage(test_df)\n\n# 2. Feature Engineering (More Advanced/Time-series focused)\ndef engineer_features_advanced(df):\n    df['bid_ask_spread'] = df['ask_qty'] - df['bid_qty']\n    df['total_qty'] = df['bid_qty'] + df['ask_qty']\n    df['buy_sell_diff'] = df['buy_qty'] - df['sell_qty']\n    df['volume_per_total_qty'] = df['volume'] / (df['total_qty'] + 1e-6)\n\n    # Introduce lag features (common in time series)\n    # Be careful with memory here, many lags can explode features\n    for col in ['bid_qty', 'ask_qty', 'volume', 'buy_qty', 'sell_qty']:\n        df[f'{col}_lag1'] = df[col].shift(1)\n        df[f'{col}_lag2'] = df[col].shift(2)\n        # Rolling features\n        df[f'{col}_rolling_mean_5'] = df[col].rolling(window=5).mean()\n        df[f'{col}_rolling_std_5'] = df[col].rolling(window=5).std()\n    \n    # Interaction features\n    df['bid_qty_x_ask_qty'] = df['bid_qty'] * df['ask_qty']\n\n    df.fillna(0, inplace=True) # Or use other imputation strategies like median/mean\n    return df\n\nprint(\"Engineering advanced features...\")\ntrain_df = engineer_features_advanced(train_df)\ntest_df = engineer_features_advanced(test_df)\nprint(\"Advanced features engineered.\")\n\n# 3. Define Features and Target (same as Level 1, ensure alignment)\nTARGET = 'label'\nfeatures = [col for col in train_df.columns if col not in ['ID', 'timestamp', TARGET]]\n\nnunique_cols = [col for col in train_df.columns if train_df[col].nunique() == 1]\ntrain_df.drop(columns=nunique_cols, inplace=True)\nfeatures = [f for f in features if f not in nunique_cols]\nif TARGET in test_df.columns:\n    test_df.drop(columns=[TARGET] + nunique_cols, inplace=True)\nelse:\n    test_df.drop(columns=nunique_cols, inplace=True)\n\ntrain_cols = set(train_df.columns)\ntest_cols = set(test_df.columns)\ncommon_features = list(train_cols.intersection(test_cols) - {TARGET, 'ID', 'timestamp'})\nfeatures = [f for f in features if f in common_features]\n\nX = train_df[features]\ny = train_df[TARGET]\n\ndel train_df # Free up memory\ngc.collect()\n\n# 4. Time Series Split\nsplit_index = int(len(X) * 0.8)\nX_train, X_val = X.iloc[:split_index], X.iloc[split_index:]\ny_train, y_val = y.iloc[:split_index], y.iloc[split_index:]\n\ndel X, y # Free up memory\ngc.collect()\n\nprint(f\"Train data shape: {X_train.shape}\")\nprint(f\"Validation data shape: {X_val.shape}\")\n\n# 5. LightGBM Model Training (GPU enabled)\nprint(\"Training LightGBM model (GPU)...\")\nlgb_params = {\n    'objective': 'regression_l1',\n    'metric': 'rmse',\n    'n_estimators': 2000, # Increased estimators for potentially better performance\n    'learning_rate': 0.02, # Reduced learning rate with more estimators\n    'feature_fraction': 0.7,\n    'bagging_fraction': 0.7,\n    'bagging_freq': 1,\n    'lambda_l1': 0.1,\n    'lambda_l2': 0.1,\n    'num_leaves': 63, # Increased complexity\n    'verbose': -1,\n    'n_jobs': -1,\n    'seed': 42,\n    'boosting_type': 'gbdt',\n    'device': 'gpu', # <<< Enable GPU\n    'gpu_platform_id': 0, # Usually 0 for single GPU\n    'gpu_device_id': 0,   # Usually 0 for single GPU\n    'max_bin': 63, # Recommended for GPU for speedup\n    'gpu_use_dp': False # Use single precision for better performance on most NVIDIA consumer GPUs\n}\n\nmodel = lgb.LGBMRegressor(**lgb_params)\n\nmodel.fit(X_train, y_train,\n          eval_set=[(X_val, y_val)],\n          eval_metric='rmse',\n          callbacks=[lgb.early_stopping(200, verbose=False)],\n          )\nprint(\"LightGBM GPU training complete.\")\n\n# 6. Prediction and Evaluation (same as Level 1)\nval_preds = model.predict(X_val)\ncorrelation, _ = pearsonr(val_preds, y_val)\nprint(f\"Validation Pearson Correlation: {correlation:.4f}\")\n\n# 7. Generate Test Predictions\nprint(\"Generating test predictions...\")\ntest_preds = model.predict(test_df[features])\ntest_preds = np.clip(test_preds, -5.0, 5.0)\n\nsubmission_df = pd.DataFrame({'ID': sample_submission['ID'], 'prediction': test_preds})\nsubmission_df.to_csv('submission2.csv', index=False)\nprint(\"Submission file created: submission2.csv\")\n\ndel X_train, X_val, y_train, y_val, test_df, model, test_preds # Clean up memory\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T03:01:52.179279Z","iopub.execute_input":"2025-07-02T03:01:52.179563Z","iopub.status.idle":"2025-07-02T03:04:03.174402Z","shell.execute_reply.started":"2025-07-02T03:01:52.179543Z","shell.execute_reply":"2025-07-02T03:04:03.173778Z"}},"outputs":[],"execution_count":null}]}