{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Step 1: Libraries\nimport pandas as pd\nimport numpy as np\nfrom xgboost import XGBRegressor\nfrom sklearn.metrics import mean_squared_error, r2_score\nfrom scipy.stats import pearsonr\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T10:46:38.001203Z","iopub.execute_input":"2025-07-23T10:46:38.001575Z","iopub.status.idle":"2025-07-23T10:46:38.008115Z","shell.execute_reply.started":"2025-07-23T10:46:38.001547Z","shell.execute_reply":"2025-07-23T10:46:38.006394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 2: Load Data\ntrain = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\ntest = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\nsample_submission = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\n\nprint(\"Train shape:\", train.shape)\nprint(\"Test shape:\", test.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T10:46:38.010517Z","iopub.execute_input":"2025-07-23T10:46:38.011837Z","iopub.status.idle":"2025-07-23T10:47:04.034661Z","shell.execute_reply.started":"2025-07-23T10:46:38.011778Z","shell.execute_reply":"2025-07-23T10:47:04.033528Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 3: Select Features\nfeatures = [col for col in train.columns if col not in ['label', 'timestamp', 'id']]\nX = train[features]\ny = train['label']\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T10:47:04.037846Z","iopub.execute_input":"2025-07-23T10:47:04.038332Z","iopub.status.idle":"2025-07-23T10:47:07.555043Z","shell.execute_reply.started":"2025-07-23T10:47:04.038297Z","shell.execute_reply":"2025-07-23T10:47:07.553623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 4: Time-Based Split (last 10% for validation)\nsplit_idx = int(len(train) * 0.9)\nX_train, X_val = X.iloc[:split_idx], X.iloc[split_idx:]\ny_train, y_val = y.iloc[:split_idx], y.iloc[split_idx:]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T10:47:07.557945Z","iopub.execute_input":"2025-07-23T10:47:07.558275Z","iopub.status.idle":"2025-07-23T10:47:07.911352Z","shell.execute_reply.started":"2025-07-23T10:47:07.558251Z","shell.execute_reply":"2025-07-23T10:47:07.910208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 5: Initialize Model\nmodel = XGBRegressor(\n    objective='reg:squarederror',\n    learning_rate=0.01,\n    max_depth=6,\n    n_estimators=1000,\n    subsample=0.8,\n    colsample_bytree=0.8,\n    random_state=42,\n    tree_method='hist'  # faster on Kaggle\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T10:47:16.730461Z","iopub.execute_input":"2025-07-23T10:47:16.730768Z","iopub.status.idle":"2025-07-23T10:47:16.736209Z","shell.execute_reply.started":"2025-07-23T10:47:16.730748Z","shell.execute_reply":"2025-07-23T10:47:16.735074Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 6: Train with Early Stopping\nmodel.fit(\n    X_train,\n    y_train,\n    eval_set=[(X_val, y_val)],\n    eval_metric='rmse',\n    early_stopping_rounds=50,\n    verbose=100\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T10:47:26.136985Z","iopub.execute_input":"2025-07-23T10:47:26.137280Z","iopub.status.idle":"2025-07-23T10:50:13.492917Z","shell.execute_reply.started":"2025-07-23T10:47:26.137261Z","shell.execute_reply":"2025-07-23T10:50:13.491738Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 7: Validation Results\nval_preds = model.predict(X_val)\nrmse = mean_squared_error(y_val, val_preds, squared=False)\nr2 = r2_score(y_val, val_preds)\ncorr, _ = pearsonr(y_val, val_preds)\n\nprint(f\"\\n📊 Validation RMSE: {rmse:.5f}\")\nprint(f\"📊 R2 Score: {r2:.5f}\")\nprint(f\"📊 Pearson Correlation: {corr:.5f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T10:50:13.494372Z","iopub.execute_input":"2025-07-23T10:50:13.494703Z","iopub.status.idle":"2025-07-23T10:50:13.932248Z","shell.execute_reply.started":"2025-07-23T10:50:13.494676Z","shell.execute_reply":"2025-07-23T10:50:13.931175Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 8: Test Prediction\nX_test = test[features]\ntest_preds = model.predict(X_test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T10:50:20.102828Z","iopub.execute_input":"2025-07-23T10:50:20.104075Z","iopub.status.idle":"2025-07-23T10:50:25.217662Z","shell.execute_reply.started":"2025-07-23T10:50:20.104029Z","shell.execute_reply":"2025-07-23T10:50:25.216874Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 9: Create Submission File\nsubmission = sample_submission.copy()\nsubmission['label'] = test_preds\nsubmission.to_csv(\"submission.csv\", index=False)\nprint(\"✅ 'submission.csv' ready for upload.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T10:50:31.709207Z","iopub.execute_input":"2025-07-23T10:50:31.709590Z","iopub.status.idle":"2025-07-23T10:50:33.654409Z","shell.execute_reply.started":"2025-07-23T10:50:31.709567Z","shell.execute_reply":"2025-07-23T10:50:33.653082Z"}},"outputs":[],"execution_count":null}]}