{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# Import pandas for data manipulation\nimport pandas as pd\n# Import numpy for numerical operations\nimport numpy as np\n# Import time for measuring preprocessing duration\nimport time\n# Import XGBRegressor for modeling\nfrom xgboost import XGBRegressor\n# Import train_test_split for validation split\nfrom sklearn.model_selection import train_test_split\n# Import mean_squared_error for evaluation\nfrom sklearn.metrics import mean_squared_error\n# Import os for file system operations\nimport os\n# Iterate through directories and files in /kaggle/input\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    # Loop through each filename\n    for filename in filenames:\n        # Print full file path\n        print(os.path.join(dirname, filename))\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-02T17:46:15.026261Z","iopub.execute_input":"2025-07-02T17:46:15.027433Z","iopub.status.idle":"2025-07-02T17:46:15.041095Z","shell.execute_reply.started":"2025-07-02T17:46:15.027399Z","shell.execute_reply":"2025-07-02T17:46:15.039889Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Load data\ntrain_data = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\ntest_data = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\n\n# Preview the structure\nprint(\"Train shape:\", train_data.shape)\nprint(\"Test shape:\", test_data.shape)\ntrain_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:45:35.268605Z","iopub.execute_input":"2025-07-06T21:45:35.268909Z","iopub.status.idle":"2025-07-06T21:46:43.753454Z","shell.execute_reply.started":"2025-07-06T21:45:35.268886Z","shell.execute_reply":"2025-07-06T21:46:43.752244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nimport matplotlib.pyplot as plt\nimport numpy as np\n\n# Suppress the RuntimeWarning\nnp.seterr(invalid='ignore')\n\n# Clip and plot the label distribution\nplt.hist(train_data['label'].clip(lower=-5, upper=5), bins=50)\nplt.title('Clipped Distribution of Label')\nplt.xlabel('Label Value')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T17:36:42.619910Z","iopub.status.idle":"2025-07-02T17:36:42.620349Z","shell.execute_reply.started":"2025-07-02T17:36:42.620133Z","shell.execute_reply":"2025-07-02T17:36:42.620152Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = train_data.reset_index()  # Ensure timestamp is a column\ntrain_data['lag_1'] = train_data['label'].shift(1)  # Previous label value\nprint(train_data[['timestamp', 'label', 'lag_1']].head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T17:36:42.621630Z","iopub.status.idle":"2025-07-02T17:36:42.622026Z","shell.execute_reply.started":"2025-07-02T17:36:42.621840Z","shell.execute_reply":"2025-07-02T17:36:42.621858Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Remove existing 'level_0' if present, then reset the index to ensure 'timestamp' is a column\nimport numpy as np\nnp.seterr(invalid='ignore')\n\nif 'level_0' in train_data.columns:\n    train_data = train_data.drop(columns=['level_0'])\ntrain_data = train_data.reset_index()\n\n# Create a lag feature by shifting the 'label' column down by one row to use the previous value as a predictor\ntrain_data['lag_1'] = train_data['label'].shift(1)\n\n# Display the first few rows of 'timestamp', 'label', and 'lag_1' to verify the lag feature\nprint(train_data[['timestamp', 'label', 'lag_1']].head())\n\n# Select key features ('bid_qty', 'ask_qty', 'volume', 'lag_1', 'label') and remove rows with NaN values (from the first lag)\nX = train_data[['bid_qty', 'ask_qty', 'volume', 'lag_1', 'label']].dropna()\n\n# Extract the 'label' column from X as the target variable y, removing it from the feature set\ny = X.pop('label')\n\n# Display the first few rows of the feature set X to confirm the structure\nprint(X.head())\n\n# Display the first few values of the target variable y to verify it matches X\nprint(y.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T17:36:42.622671Z","iopub.status.idle":"2025-07-02T17:36:42.623018Z","shell.execute_reply.started":"2025-07-02T17:36:42.622842Z","shell.execute_reply":"2025-07-02T17:36:42.622860Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nnp.seterr(invalid='ignore')\n# Adding more from competition data\nX_expanded = train_data[['bid_qty', 'ask_qty', 'volume', 'lag_1', 'label', 'X1', 'X2', 'X3']].dropna()\ny_expanded = X_expanded.pop('label')\nfrom sklearn.model_selection import train_test_split\nX_train_exp, X_val_exp, y_train_exp, y_val_exp = train_test_split(X_expanded, y_expanded, test_size=0.2, shuffle=False)\nfrom xgboost import XGBRegressor\nmodel_exp = XGBRegressor()\nmodel_exp.fit(X_train_exp, y_train_exp)\nfrom scipy.stats import pearsonr\npredictions = model_exp.predict(X_val_exp)\ncorr, _ = pearsonr(predictions, y_val_exp)\nprint(\"Expanded Pearson correlation:\", corr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T17:36:42.624287Z","iopub.status.idle":"2025-07-02T17:36:42.624657Z","shell.execute_reply.started":"2025-07-02T17:36:42.624467Z","shell.execute_reply":"2025-07-02T17:36:42.624483Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np \nnp.seterr(invalid='ignore')\nfrom sklearn.model_selection import GridSearchCV\nfrom xgboost import XGBRegressor\n\n#Test Robot Settings \nparam_grid = {'n_estimators': [100, 200], 'max_depth': [3, 5], 'learning_rate': [0.01, 0.1]}\nmodel = XGBRegressor()\ngrid_search = GridSearchCV(model, param_grid, cv=3, scoring='r2')  # Note: Use Pearson if possible\ngrid_search.fit(X_train_exp, y_train_exp)\n\n#Best Settings and Score\nprint(\"Best parameters:\", grid_search.best_params_)\nbest_model = grid_search.best_estimator_\npredictions = best_model.predict(X_val_exp)\nfrom scipy.stats import pearsonr\ncorr, _ = pearsonr(predictions, y_val_exp)\nprint(\"Tuned Pearson correlation:\", corr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T17:36:42.625287Z","iopub.status.idle":"2025-07-02T17:36:42.625634Z","shell.execute_reply.started":"2025-07-02T17:36:42.625462Z","shell.execute_reply":"2025-07-02T17:36:42.625477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nnp.seterr(invalid='ignore')\nfrom sklearn.ensemble import VotingRegressor\nfrom xgboost import XGBRegressor\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.model_selection import train_test_split\nfrom scipy.stats import pearsonr\n\n# Build two robots\nxgb_model = XGBRegressor(learning_rate=0.1, max_depth=3, n_estimators=100)  # Use tuned params\nlin_model = LinearRegression()\n\n# Make a super team\nensemble = VotingRegressor([('xgb', xgb_model), ('lin', lin_model)])\nensemble.fit(X_train_exp, y_train_exp)\n\n# Check super team score\npredictions = ensemble.predict(X_val_exp)\ncorr, _ = pearsonr(predictions, y_val_exp)\nprint(\"Ensemble Pearson correlation:\", corr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T17:36:42.627133Z","iopub.status.idle":"2025-07-02T17:36:42.627407Z","shell.execute_reply.started":"2025-07-02T17:36:42.627271Z","shell.execute_reply":"2025-07-02T17:36:42.627282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nnp.seterr(invalid='ignore')\n\n# Take a smaller sample to save resources\ntrain_data = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet').sample(frac=0.1, random_state=42)\nprint(train_data.shape)\ntest_data = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\nprint(test_data.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T17:36:42.628292Z","iopub.status.idle":"2025-07-02T17:36:42.628653Z","shell.execute_reply.started":"2025-07-02T17:36:42.628473Z","shell.execute_reply":"2025-07-02T17:36:42.628489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}