{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-20T13:13:16.443683Z","iopub.execute_input":"2025-07-20T13:13:16.443841Z","iopub.status.idle":"2025-07-20T13:13:17.605166Z","shell.execute_reply.started":"2025-07-20T13:13:16.443825Z","shell.execute_reply":"2025-07-20T13:13:17.604294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 1: Imports and Setup\nimport pandas as pd\nimport numpy as np\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.decomposition import PCA\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport gc\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n# Config\nTRAIN_PATH = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\nTEST_PATH = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\nSAMPLE_SUB_PATH = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\nLABEL_COL = \"label\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-20T13:13:25.965532Z","iopub.execute_input":"2025-07-20T13:13:25.965806Z","iopub.status.idle":"2025-07-20T13:13:27.222476Z","shell.execute_reply.started":"2025-07-20T13:13:25.965782Z","shell.execute_reply":"2025-07-20T13:13:27.221930Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 2: Load Data\nprint(\"Loading data...\")\ntrain_df = pd.read_parquet(TRAIN_PATH)\ntest_df = pd.read_parquet(TEST_PATH)\nsample_df = pd.read_csv(SAMPLE_SUB_PATH)\n\nprint(f\"Train shape: {train_df.shape}\")\nprint(f\"Test shape: {test_df.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-20T13:13:31.799104Z","iopub.execute_input":"2025-07-20T13:13:31.799494Z","iopub.status.idle":"2025-07-20T13:14:13.679834Z","shell.execute_reply.started":"2025-07-20T13:13:31.799472Z","shell.execute_reply":"2025-07-20T13:14:13.679094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop timestamp/index columns if present\nif '__index_level_0__' in train_df.columns:\n    train_df = train_df.drop(columns=['__index_level_0__'])\n\n# Confirm label exists\nassert LABEL_COL in train_df.columns, f\"'{LABEL_COL}' column missing from train\"\n\n# Separate features and target\nX = train_df.drop(columns=[LABEL_COL])\ny = train_df[LABEL_COL]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-20T13:14:29.755594Z","iopub.execute_input":"2025-07-20T13:14:29.756402Z","iopub.status.idle":"2025-07-20T13:14:30.715204Z","shell.execute_reply.started":"2025-07-20T13:14:29.756377Z","shell.execute_reply":"2025-07-20T13:14:30.714428Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Apply PCA to reduce from ~785 columns to e.g., 100 principal components\nprint(\"Applying PCA...\")\npca = PCA(n_components=100, random_state=42)\nX_pca = pca.fit_transform(X)\nX_test_pca = pca.transform(test_df[X.columns])\n\nprint(f\"PCA shape: {X_pca.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-20T13:14:44.365306Z","iopub.execute_input":"2025-07-20T13:14:44.365573Z","iopub.status.idle":"2025-07-20T13:15:24.354630Z","shell.execute_reply.started":"2025-07-20T13:14:44.365551Z","shell.execute_reply":"2025-07-20T13:15:24.354007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from xgboost import XGBRegressor\n\n# Train/Validation split remains the same\nX_train, X_val, y_train, y_val = train_test_split(X_pca, y, test_size=0.2, random_state=42)\n\n# Train XGBoost Regressor\nprint(\"Training XGBoost Regressor...\")\nxgb_model = XGBRegressor(\n    n_estimators=300,\n    max_depth=8,\n    learning_rate=0.05,\n    subsample=0.8,\n    colsample_bytree=0.8,\n    reg_alpha=0.5,\n    reg_lambda=0.5,\n    random_state=42,\n    n_jobs=-1,\n    tree_method='hist'  # Faster for large datasets\n)\nxgb_model.fit(X_train, y_train)\n\n# Validation Score\ny_pred = xgb_model.predict(X_val)\nmse = mean_squared_error(y_val, y_pred)\nprint(f\"Validation MSE (XGB): {mse:.5f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-20T13:39:33.875656Z","iopub.execute_input":"2025-07-20T13:39:33.876374Z","iopub.status.idle":"2025-07-20T13:40:18.991263Z","shell.execute_reply.started":"2025-07-20T13:39:33.876348Z","shell.execute_reply":"2025-07-20T13:40:18.990617Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Generating predictions...\")\ntest_preds = xgb_model.predict(X_test_pca)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-20T13:43:18.138621Z","iopub.execute_input":"2025-07-20T13:43:18.139305Z","iopub.status.idle":"2025-07-20T13:43:20.598955Z","shell.execute_reply.started":"2025-07-20T13:43:18.139279Z","shell.execute_reply":"2025-07-20T13:43:20.598301Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df[\"prediction\"] = test_preds\nsample_df.to_csv(\"submission_xgb.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-20T13:44:32.278376Z","iopub.execute_input":"2025-07-20T13:44:32.279123Z","iopub.status.idle":"2025-07-20T13:44:33.034807Z","shell.execute_reply.started":"2025-07-20T13:44:32.279095Z","shell.execute_reply":"2025-07-20T13:44:33.034051Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}