{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":10211152,"sourceType":"datasetVersion","datasetId":6311079}],"dockerImageVersionId":30805,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:37:21.171383Z","iopub.execute_input":"2024-12-17T00:37:21.171697Z","iopub.status.idle":"2024-12-17T00:37:22.328989Z","shell.execute_reply.started":"2024-12-17T00:37:21.171670Z","shell.execute_reply":"2024-12-17T00:37:22.327889Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import necessary libraries\nimport os\nimport numpy as np\nimport pandas as pd\nimport gc\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import r2_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:37:22.330835Z","iopub.execute_input":"2024-12-17T00:37:22.331344Z","iopub.status.idle":"2024-12-17T00:37:23.570878Z","shell.execute_reply.started":"2024-12-17T00:37:22.331304Z","shell.execute_reply":"2024-12-17T00:37:23.569856Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Enable GPU in Kaggle (you need to select GPU in settings manually)\nprint(\"Checking GPU availability...\")\n!nvidia-smi","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:37:23.572114Z","iopub.execute_input":"2024-12-17T00:37:23.572481Z","iopub.status.idle":"2024-12-17T00:37:24.653897Z","shell.execute_reply.started":"2024-12-17T00:37:23.572453Z","shell.execute_reply":"2024-12-17T00:37:24.652859Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Path to the directory containing Parquet files\npath = \"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet\"\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:37:24.656068Z","iopub.execute_input":"2024-12-17T00:37:24.656376Z","iopub.status.idle":"2024-12-17T00:37:24.660695Z","shell.execute_reply.started":"2024-12-17T00:37:24.656348Z","shell.execute_reply":"2024-12-17T00:37:24.659620Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Output file to store the combined data\noutput_file = \"processed_train_subset.parquet\"\n\n# Initialize an empty DataFrame to store the combined data\ncombined_data = pd.DataFrame()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:37:24.661706Z","iopub.execute_input":"2024-12-17T00:37:24.661988Z","iopub.status.idle":"2024-12-17T00:37:24.676773Z","shell.execute_reply.started":"2024-12-17T00:37:24.661963Z","shell.execute_reply":"2024-12-17T00:37:24.676078Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Process and sample the dataset\nprint(\"Processing data...\")\nfor part in os.listdir(path):\n    file_path = os.path.join(path, part)\n    print(f\"Processing: {file_path}\")\n    \n    # Load the Parquet file\n    subset = pd.read_parquet(file_path)\n    \n    # Take a 20% random sample of the data for faster computation\n    sampled_subset = subset.sample(frac=0.20, random_state=42)\n    \n    # Concatenate the sampled data into the combined DataFrame\n    combined_data = pd.concat([combined_data, sampled_subset], ignore_index=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:37:24.677724Z","iopub.execute_input":"2024-12-17T00:37:24.678023Z","iopub.status.idle":"2024-12-17T00:38:12.571701Z","shell.execute_reply.started":"2024-12-17T00:37:24.677999Z","shell.execute_reply":"2024-12-17T00:38:12.570965Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" # Clean up memory after each part is processed\ndel subset\ngc.collect()\n\n# Save the combined subset of data to a single Parquet file\ncombined_data.to_parquet(output_file, index=False, engine=\"pyarrow\")\nprint(f\"Processed data saved to: {output_file}\")\n\n# Load the processed Parquet file\nprocessed_df = pd.read_parquet(\"processed_train_subset.parquet\")\n\n# Check the structure of the dataset\nprint(processed_df.info())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:38:12.572880Z","iopub.execute_input":"2024-12-17T00:38:12.573193Z","iopub.status.idle":"2024-12-17T00:38:51.362313Z","shell.execute_reply.started":"2024-12-17T00:38:12.573167Z","shell.execute_reply":"2024-12-17T00:38:51.361415Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Correlation analysis in chunks\nchunk_size = 100_000  # Adjusted for better performance\nfeatures = [col for col in processed_df.columns if 'feature' in col]\ntarget = 'responder_6'\ncorrelations = None\n\nfor i in range(0, len(processed_df), chunk_size):\n    chunk = processed_df.iloc[i:i + chunk_size]\n    # Compute correlations only for features\n    chunk_corr = chunk[features + [target]].corr()[target].drop(target)\n    \n    if correlations is None:\n        correlations = chunk_corr\n    else:\n        correlations = (correlations + chunk_corr) / 2\n\n# Sort and display top correlated features\ncorrelations = correlations.sort_values(ascending=False)\nprint(\"Top 10 Features Correlated with responder_6:\")\nprint(correlations.head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:38:51.363673Z","iopub.execute_input":"2024-12-17T00:38:51.364103Z","iopub.status.idle":"2024-12-17T00:41:12.715157Z","shell.execute_reply.started":"2024-12-17T00:38:51.364059Z","shell.execute_reply":"2024-12-17T00:41:12.714285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a filtered dataset with top features and target\ntop_features = correlations.head(10).index.tolist()\nfiltered_df = processed_df[top_features + ['responder_6']].copy()\n\n# Drop rows with missing values\nfiltered_df.dropna(inplace=True)\nprint(\"After dropping missing values:\")\nprint(filtered_df.info())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:41:12.716350Z","iopub.execute_input":"2024-12-17T00:41:12.716625Z","iopub.status.idle":"2024-12-17T00:41:13.442855Z","shell.execute_reply.started":"2024-12-17T00:41:12.716599Z","shell.execute_reply":"2024-12-17T00:41:13.442025Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define features (X) and target (y)\nX = filtered_df[top_features]\ny = filtered_df['responder_6']\n\n# Split into training and validation sets\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:41:13.446537Z","iopub.execute_input":"2024-12-17T00:41:13.446883Z","iopub.status.idle":"2024-12-17T00:41:15.017706Z","shell.execute_reply.started":"2024-12-17T00:41:13.446856Z","shell.execute_reply":"2024-12-17T00:41:15.016776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import xgboost as xgb\n\n# Train an XGBoost model with GPU\nprint(\"Training XGBoost...\")\nxgb_model = xgb.XGBRegressor(tree_method=\"hist\", device=\"cuda\", n_estimators=100, learning_rate=0.1, random_state=42)\nxgb_model.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:41:15.018921Z","iopub.execute_input":"2024-12-17T00:41:15.019299Z","iopub.status.idle":"2024-12-17T00:41:24.879041Z","shell.execute_reply.started":"2024-12-17T00:41:15.019258Z","shell.execute_reply":"2024-12-17T00:41:24.878194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predict on the validation set\ny_pred_xgb = xgb_model.predict(X_val)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:41:24.880107Z","iopub.execute_input":"2024-12-17T00:41:24.880357Z","iopub.status.idle":"2024-12-17T00:41:25.211943Z","shell.execute_reply.started":"2024-12-17T00:41:24.880334Z","shell.execute_reply":"2024-12-17T00:41:25.211110Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import r2_score, mean_squared_error\n\n# R² Score\nxgb_r2 = r2_score(y_val, y_pred_xgb)\nprint(f\"XGBoost R² Score: {xgb_r2}\")\n\n# Mean Squared Error\nxgb_mse = mean_squared_error(y_val, y_pred_xgb)\nprint(f\"XGBoost Mean Squared Error: {xgb_mse}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:41:25.213007Z","iopub.execute_input":"2024-12-17T00:41:25.213264Z","iopub.status.idle":"2024-12-17T00:41:25.230296Z","shell.execute_reply.started":"2024-12-17T00:41:25.213239Z","shell.execute_reply":"2024-12-17T00:41:25.229545Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save the trained model\nMODEL_PATH = \"final_xgb_model.json\"  # Name and path of the model\nxgb_model.save_model(MODEL_PATH)\nprint(f\"Model saved to {MODEL_PATH}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:41:25.231227Z","iopub.execute_input":"2024-12-17T00:41:25.231525Z","iopub.status.idle":"2024-12-17T00:41:25.250669Z","shell.execute_reply.started":"2024-12-17T00:41:25.231497Z","shell.execute_reply":"2024-12-17T00:41:25.249044Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import lightgbm as lgb\n\n# Train a LightGBM model with GPU\nprint(\"Training LightGBM...\")\nlgb_model = lgb.LGBMRegressor(device='gpu', n_estimators=100, learning_rate=0.1, random_state=42)\nlgb_model.fit(X_train, y_train)\n\n# Predict and evaluate LightGBM model\ny_pred_lgb = lgb_model.predict(X_val)\nlgb_r2 = r2_score(y_val, y_pred_lgb)\nprint(f\"LightGBM R² Score: {lgb_r2}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:41:25.251859Z","iopub.execute_input":"2024-12-17T00:41:25.252237Z","iopub.status.idle":"2024-12-17T00:41:49.362694Z","shell.execute_reply.started":"2024-12-17T00:41:25.252197Z","shell.execute_reply":"2024-12-17T00:41:49.361745Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from catboost import CatBoostRegressor\n\n# Train CatBoost model\nprint(\"Training CatBoost...\")\ncat_model = CatBoostRegressor(iterations=100, learning_rate=0.1, depth=6, task_type=\"GPU\", random_seed=42, verbose=0)\ncat_model.fit(X_train, y_train)\n\n# Predict and evaluate CatBoost model\ny_pred_cat = cat_model.predict(X_val)\ncat_r2 = r2_score(y_val, y_pred_cat)\nprint(f\"CatBoost R² Score: {cat_r2}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:41:49.364225Z","iopub.execute_input":"2024-12-17T00:41:49.365236Z","iopub.status.idle":"2024-12-17T00:42:05.570609Z","shell.execute_reply.started":"2024-12-17T00:41:49.365189Z","shell.execute_reply":"2024-12-17T00:42:05.569619Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import r2_score, mean_squared_error\n\n# Train a Linear Regression model\nprint(\"Training Linear Regression...\")\nlinear_model = LinearRegression()\nlinear_model.fit(X_train, y_train)\n\n# Predict and evaluate\ny_pred_linear = linear_model.predict(X_val)\nlinear_r2 = r2_score(y_val, y_pred_linear)\nlinear_mse = mean_squared_error(y_val, y_pred_linear)\n\n# Display results\nprint(f\"Linear Regression R² Score: {linear_r2}\")\nprint(f\"Linear Regression Mean Squared Error: {linear_mse}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:42:05.571780Z","iopub.execute_input":"2024-12-17T00:42:05.572115Z","iopub.status.idle":"2024-12-17T00:42:08.104374Z","shell.execute_reply.started":"2024-12-17T00:42:05.572088Z","shell.execute_reply":"2024-12-17T00:42:08.102581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.hist(y, bins=50, color='blue', alpha=0.7)\nplt.title(\"Target Variable Distribution\")\nplt.xlabel(\"Responder_6\")\nplt.ylabel(\"Frequency\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:42:08.106022Z","iopub.execute_input":"2024-12-17T00:42:08.106470Z","iopub.status.idle":"2024-12-17T00:42:08.562225Z","shell.execute_reply.started":"2024-12-17T00:42:08.106420Z","shell.execute_reply":"2024-12-17T00:42:08.561384Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\n\nsns.heatmap(filtered_df.corr(), annot=True, fmt=\".2f\", cmap=\"coolwarm\")\nplt.title(\"Correlation Heatmap of Selected Features\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:42:08.563412Z","iopub.execute_input":"2024-12-17T00:42:08.563777Z","iopub.status.idle":"2024-12-17T00:42:12.323970Z","shell.execute_reply.started":"2024-12-17T00:42:08.563739Z","shell.execute_reply":"2024-12-17T00:42:12.323094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for lag in range(1, 4):\n    X[f'feature_16_lag_{lag}'] = X['feature_16'].shift(lag)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:42:12.325058Z","iopub.execute_input":"2024-12-17T00:42:12.325543Z","iopub.status.idle":"2024-12-17T00:42:12.377387Z","shell.execute_reply.started":"2024-12-17T00:42:12.325513Z","shell.execute_reply":"2024-12-17T00:42:12.376720Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X['feature_16_roll_mean'] = X['feature_16'].rolling(window=5).mean()\nX['feature_16_roll_std'] = X['feature_16'].rolling(window=5).std()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:42:12.378360Z","iopub.execute_input":"2024-12-17T00:42:12.378701Z","iopub.status.idle":"2024-12-17T00:42:12.996387Z","shell.execute_reply.started":"2024-12-17T00:42:12.378664Z","shell.execute_reply":"2024-12-17T00:42:12.995636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X['feature_16_17_interaction'] = X['feature_16'] * X['feature_17']\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:42:12.997437Z","iopub.execute_input":"2024-12-17T00:42:12.997701Z","iopub.status.idle":"2024-12-17T00:42:13.021453Z","shell.execute_reply.started":"2024-12-17T00:42:12.997676Z","shell.execute_reply":"2024-12-17T00:42:13.020853Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# Feature importance for XGBoost\nxgb_feature_importance = xgb_model.feature_importances_\nsorted_idx = np.argsort(xgb_feature_importance)\n\nplt.barh(np.array(X_train.columns)[sorted_idx], xgb_feature_importance[sorted_idx])\nplt.xlabel(\"Feature Importance\")\nplt.title(\"XGBoost Feature Importance\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:42:13.022295Z","iopub.execute_input":"2024-12-17T00:42:13.022502Z","iopub.status.idle":"2024-12-17T00:42:13.268948Z","shell.execute_reply.started":"2024-12-17T00:42:13.022481Z","shell.execute_reply":"2024-12-17T00:42:13.268128Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"top_features = ['feature_16', 'feature_17', 'feature_60', 'feature_51', 'feature_15','feature_58','feature_34']\nX_train_top = X_train[top_features]\nX_val_top = X_val[top_features]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:42:13.270076Z","iopub.execute_input":"2024-12-17T00:42:13.270727Z","iopub.status.idle":"2024-12-17T00:42:13.360185Z","shell.execute_reply.started":"2024-12-17T00:42:13.270688Z","shell.execute_reply":"2024-12-17T00:42:13.359445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Subsample for faster training\nX_train_sample = X_train_top.sample(frac=0.2, random_state=42)  # Use 20% of the training data\ny_train_sample = y_train.sample(frac=0.2, random_state=42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:42:13.361355Z","iopub.execute_input":"2024-12-17T00:42:13.361993Z","iopub.status.idle":"2024-12-17T00:42:13.964312Z","shell.execute_reply.started":"2024-12-17T00:42:13.361951Z","shell.execute_reply":"2024-12-17T00:42:13.963312Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from xgboost import XGBRegressor\nfrom sklearn.model_selection import RandomizedSearchCV\n\n# Define the parameter distribution\nparam_dist = {\n    'max_depth': [4, 6, 8],\n    'learning_rate': [0.01, 0.1, 0.2],\n    'n_estimators': [50, 100, 200],\n    'subsample': [0.8, 1.0],\n    'colsample_bytree': [0.8, 1.0]\n}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:42:13.965759Z","iopub.execute_input":"2024-12-17T00:42:13.966521Z","iopub.status.idle":"2024-12-17T00:42:13.971481Z","shell.execute_reply.started":"2024-12-17T00:42:13.966483Z","shell.execute_reply":"2024-12-17T00:42:13.970672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize RandomizedSearchCV\nrandom_search = RandomizedSearchCV(\n    estimator=XGBRegressor(tree_method='hist', device = 'cuda', random_state=42),\n    param_distributions=param_dist,\n    n_iter=20,  # Number of parameter combinations to try\n    scoring='r2',  # Evaluation metric\n    cv=3,  # Number of cross-validation folds\n    verbose=1,\n    random_state=42\n)\n\n# Run the search\nrandom_search.fit(X_train_sample, y_train_sample)  # Use the subsampled data\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:42:13.972839Z","iopub.execute_input":"2024-12-17T00:42:13.973592Z","iopub.status.idle":"2024-12-17T00:43:43.062278Z","shell.execute_reply.started":"2024-12-17T00:42:13.973554Z","shell.execute_reply":"2024-12-17T00:43:43.061301Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Best Parameters: {random_search.best_params_}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:43:43.067304Z","iopub.execute_input":"2024-12-17T00:43:43.067970Z","iopub.status.idle":"2024-12-17T00:43:43.072205Z","shell.execute_reply.started":"2024-12-17T00:43:43.067943Z","shell.execute_reply":"2024-12-17T00:43:43.071203Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the final model on the full dataset\nfinal_xgb_model = XGBRegressor(\n    tree_method='hist',\n    random_state=42,\n    **random_search.best_params_\n)\n\nfinal_xgb_model.fit(X_train_top, y_train)  # Train on the full dataset with top features\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:43:43.073185Z","iopub.execute_input":"2024-12-17T00:43:43.073470Z","iopub.status.idle":"2024-12-17T00:44:32.763606Z","shell.execute_reply.started":"2024-12-17T00:43:43.073425Z","shell.execute_reply":"2024-12-17T00:44:32.762761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import r2_score, mean_squared_error\n\n# Predict on the validation set\ny_pred_final = final_xgb_model.predict(X_val_top)\n\n# Evaluate the final model\nfinal_r2 = r2_score(y_val, y_pred_final)\nfinal_mse = mean_squared_error(y_val, y_pred_final)\n\nprint(f\"Final XGBoost R² Score: {final_r2}\")\nprint(f\"Final XGBoost Mean Squared Error: {final_mse}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:44:32.764707Z","iopub.execute_input":"2024-12-17T00:44:32.765004Z","iopub.status.idle":"2024-12-17T00:44:35.896815Z","shell.execute_reply.started":"2024-12-17T00:44:32.764976Z","shell.execute_reply":"2024-12-17T00:44:35.895744Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nplt.scatter(y_val, y_pred_xgb, alpha=0.6, color='blue', label=\"XGBoost\")\nplt.plot([min(y_val), max(y_val)], [min(y_val), max(y_val)], color='red', linestyle='--', label=\"Perfect Fit\")\nplt.xlabel(\"Actual Responder_6\")\nplt.ylabel(\"Predicted Responder_6\")\nplt.title(\"Actual vs Predicted (XGBoost)\")\nplt.legend()\nplt.grid(True)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:44:35.897898Z","iopub.execute_input":"2024-12-17T00:44:35.898156Z","iopub.status.idle":"2024-12-17T00:44:51.518320Z","shell.execute_reply.started":"2024-12-17T00:44:35.898124Z","shell.execute_reply":"2024-12-17T00:44:51.517286Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Model names and R² scores\nmodel_names = ['XGBoost', 'LightGBM', 'CatBoost', 'Linear Regression']\nr2_scores = [xgb_r2, lgb_r2, cat_r2, linear_r2]\n\n# Bar plot for R² scores\nplt.figure(figsize=(8, 6))\nplt.bar(model_names, r2_scores, color=['blue', 'green', 'orange', 'red'])\nplt.title(\"Model Comparison (R² Scores)\")\nplt.xlabel(\"Models\")\nplt.ylabel(\"R² Score\")\nplt.ylim(0, max(r2_scores) + 0.01)  # Add some padding to the top\nplt.grid(axis='y')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:44:51.519480Z","iopub.execute_input":"2024-12-17T00:44:51.519764Z","iopub.status.idle":"2024-12-17T00:44:51.746455Z","shell.execute_reply.started":"2024-12-17T00:44:51.519738Z","shell.execute_reply":"2024-12-17T00:44:51.745562Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pyarrow.dataset as ds\nimport pandas as pd\nimport numpy as np\n\n# Load the test data using pyarrow.dataset\ndataset = ds.dataset(\"/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet\", format=\"parquet\")\n\n# Convert to pandas DataFrame\ntest_df = dataset.to_table().to_pandas()\n\n# Fix the column types to match the training data\ntest_df['date_id'] = test_df['date_id'].astype('int16')\ntest_df['time_id'] = test_df['time_id'].astype('int16')\ntest_df['symbol_id'] = test_df['symbol_id'].astype('int8')\ntest_df['weight'] = test_df['weight'].astype('float32')\n\nprint(\"Test data loaded and column types fixed:\")\nprint(test_df.info())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:44:51.747560Z","iopub.execute_input":"2024-12-17T00:44:51.747904Z","iopub.status.idle":"2024-12-17T00:44:51.780574Z","shell.execute_reply.started":"2024-12-17T00:44:51.747877Z","shell.execute_reply":"2024-12-17T00:44:51.779751Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Select the top features used during training\nX_test = test_df[top_features]  # Replace 'top_features' with the correct feature list\n\n# Generate predictions\ny_test_pred = final_xgb_model.predict(X_test)\n\n# Prepare submission file\nsubmission = pd.DataFrame({\n    \"id\": test_df['row_id'],  # Replace 'id' with the actual ID column from the test set\n    \"prediction\": y_test_pred\n})\n\n# Save submission file as 'submission.parquet'\nsubmission.to_parquet(\"submission.parquet\", index=False, engine=\"pyarrow\")\n\nprint(\"Submission file saved as 'submission.parquet'\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:44:51.781549Z","iopub.execute_input":"2024-12-17T00:44:51.781792Z","iopub.status.idle":"2024-12-17T00:44:51.792563Z","shell.execute_reply.started":"2024-12-17T00:44:51.781768Z","shell.execute_reply":"2024-12-17T00:44:51.790783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save the trained model\nMODEL_PATH = \"final_xgb_model.json\"  # Name and path of the model\nxgb_model.save_model(MODEL_PATH)\nprint(f\"Model saved to {MODEL_PATH}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:44:51.793470Z","iopub.execute_input":"2024-12-17T00:44:51.793691Z","iopub.status.idle":"2024-12-17T00:44:51.817739Z","shell.execute_reply.started":"2024-12-17T00:44:51.793667Z","shell.execute_reply":"2024-12-17T00:44:51.817031Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the saved XGBoost model\nxgb_model_loaded = xgb.XGBRegressor()\nxgb_model_loaded.load_model(MODEL_PATH)\n\n# Make predictions with the loaded model\ny_pred_loaded = xgb_model_loaded.predict(X_val)\n\n# Verify that predictions are the same\nprint(f\"R² Score (Reloaded Model): {r2_score(y_val, y_pred_loaded)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:44:51.819057Z","iopub.execute_input":"2024-12-17T00:44:51.819441Z","iopub.status.idle":"2024-12-17T00:44:53.205952Z","shell.execute_reply.started":"2024-12-17T00:44:51.819405Z","shell.execute_reply":"2024-12-17T00:44:53.205171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Verify training data\nprint(\"X_train shape:\", X_train.shape)\nprint(\"y_train shape:\", y_train.shape)\n\n# Train the model\nxgb_model = xgb.XGBRegressor(tree_method=\"hist\", device=\"cuda\", n_estimators=100, learning_rate=0.1, random_state=42)\nxgb_model.fit(X_train, y_train)\n\n# Predict to confirm training worked\ny_pred_xgb = xgb_model.predict(X_val)\nprint(\"R² Score:\", r2_score(y_val, y_pred_xgb))\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:44:53.206816Z","iopub.execute_input":"2024-12-17T00:44:53.207180Z","iopub.status.idle":"2024-12-17T00:45:02.469261Z","shell.execute_reply.started":"2024-12-17T00:44:53.207146Z","shell.execute_reply":"2024-12-17T00:45:02.468215Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save the trained model\nxgb_model.save_model(\"final_xgb_model.json\")\nprint(\"Model saved successfully as 'final_xgb_model.json'\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:45:02.470420Z","iopub.execute_input":"2024-12-17T00:45:02.470699Z","iopub.status.idle":"2024-12-17T00:45:02.487603Z","shell.execute_reply.started":"2024-12-17T00:45:02.470673Z","shell.execute_reply":"2024-12-17T00:45:02.486368Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the saved model\nxgb_model_loaded = xgb.XGBRegressor()\nxgb_model_loaded.load_model(\"final_xgb_model.json\")\n\n# Test predictions\ny_pred_loaded = xgb_model_loaded.predict(X_val)\nprint(\"R² Score (Reloaded Model):\", r2_score(y_val, y_pred_loaded))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:45:02.488495Z","iopub.execute_input":"2024-12-17T00:45:02.488782Z","iopub.status.idle":"2024-12-17T00:45:03.880026Z","shell.execute_reply.started":"2024-12-17T00:45:02.488758Z","shell.execute_reply":"2024-12-17T00:45:03.879240Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# List all files in the working directory\nprint(os.listdir(\"/kaggle/working/\"))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:45:03.880829Z","iopub.execute_input":"2024-12-17T00:45:03.881285Z","iopub.status.idle":"2024-12-17T00:45:03.885578Z","shell.execute_reply.started":"2024-12-17T00:45:03.881257Z","shell.execute_reply":"2024-12-17T00:45:03.884604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# Check for files in the working directory\nprint(\"Working Directory Files:\", os.listdir(\"/kaggle/working/\"))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:45:03.886649Z","iopub.execute_input":"2024-12-17T00:45:03.886915Z","iopub.status.idle":"2024-12-17T00:45:03.897455Z","shell.execute_reply.started":"2024-12-17T00:45:03.886891Z","shell.execute_reply":"2024-12-17T00:45:03.896588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import xgboost as xgb\n\n# Correct path to the uploaded model\nMODEL_PATH = \"/kaggle/working/final_xgb_model.json\"\n\n# Load the model\nxgb_model = xgb.XGBRegressor()\nxgb_model.load_model(MODEL_PATH)\nprint(\"Model loaded successfully.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:45:03.898782Z","iopub.execute_input":"2024-12-17T00:45:03.899093Z","iopub.status.idle":"2024-12-17T00:45:03.936887Z","shell.execute_reply.started":"2024-12-17T00:45:03.899068Z","shell.execute_reply":"2024-12-17T00:45:03.935431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import polars as pl\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:45:15.582531Z","iopub.execute_input":"2024-12-17T00:45:15.583310Z","iopub.status.idle":"2024-12-17T00:45:15.587281Z","shell.execute_reply.started":"2024-12-17T00:45:15.583274Z","shell.execute_reply":"2024-12-17T00:45:15.586301Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"Make a prediction using the loaded XGBoost model.\"\"\"\n    # Convert Polars DataFrame to Pandas and fill missing values\n    test_df = test.to_pandas().fillna(0)\n\n    # Select the relevant features\n    top_features = ['feature_16', 'feature_17', 'feature_51', 'feature_60',\n                    'feature_15', 'feature_58', 'feature_08', 'feature_50',\n                    'feature_68', 'feature_34']\n    X_test = test_df[top_features]\n\n    # Predict using the model\n    predictions = xgb_model.predict(X_test)\n\n    # Debug: Print the first few predictions and rows\n    print(\"Sample test rows:\")\n    print(test_df.head(3))\n    print(\"Sample predictions:\")\n    print(predictions[:5])\n\n    # Return predictions in the required format\n    return pd.DataFrame({\n        'row_id': test_df['row_id'],\n        'responder_6': predictions\n    })\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:45:16.210661Z","iopub.execute_input":"2024-12-17T00:45:16.210989Z","iopub.status.idle":"2024-12-17T00:45:16.216937Z","shell.execute_reply.started":"2024-12-17T00:45:16.210961Z","shell.execute_reply":"2024-12-17T00:45:16.215943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nsys.path.append('/kaggle/input/jane-street-real-time-market-data-forecasting/kaggle_evaluation')\nimport kaggle_evaluation.jane_street_inference_server\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:46:41.639070Z","iopub.execute_input":"2024-12-17T00:46:41.639406Z","iopub.status.idle":"2024-12-17T00:46:41.832245Z","shell.execute_reply.started":"2024-12-17T00:46:41.639374Z","shell.execute_reply":"2024-12-17T00:46:41.831556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:46:44.889388Z","iopub.execute_input":"2024-12-17T00:46:44.890113Z","iopub.status.idle":"2024-12-17T00:46:45.193001Z","shell.execute_reply.started":"2024-12-17T00:46:44.890078Z","shell.execute_reply":"2024-12-17T00:46:45.192109Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save predictions as submission.parquet\nsubmission.to_parquet(\"submission.parquet\", index=False)\nprint(\"Submission file saved as submission.parquet\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:47:04.198656Z","iopub.execute_input":"2024-12-17T00:47:04.199527Z","iopub.status.idle":"2024-12-17T00:47:04.207086Z","shell.execute_reply.started":"2024-12-17T00:47:04.199475Z","shell.execute_reply":"2024-12-17T00:47:04.206150Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}