{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84493,"databundleVersionId":9849268,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import dask.dataframe as dd\nimport lightgbm as lgb\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\n\n# Load the Parquet file in chunks using Dask\ndef load_and_sample_data(filepath, sample_fraction=0.05):\n    print(f\"Loading and sampling data from {filepath} using Dask...\")\n    \n    # Load the parquet file as a Dask dataframe\n    dask_df = dd.read_parquet(filepath, engine='pyarrow')\n    \n    # Convert a fraction of the Dask dataframe into a Pandas dataframe\n    sampled_df = dask_df.sample(frac=sample_fraction).compute()  # This loads only the sampled fraction into memory\n    return sampled_df\n\n# Load the train data and sample\ntrain_data = load_and_sample_data('/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet', sample_fraction=0.05)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-16T08:43:39.819354Z","iopub.execute_input":"2024-10-16T08:43:39.820365Z","iopub.status.idle":"2024-10-16T08:44:41.493688Z","shell.execute_reply.started":"2024-10-16T08:43:39.820292Z","shell.execute_reply":"2024-10-16T08:44:41.492808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the features (anonymized market data) and target (responder_6)\nX = train_data[[f'feature_{i:02d}' for i in range(79)]].values\ny = train_data['responder_6'].values\nweights = train_data['weight'].values","metadata":{"execution":{"iopub.status.busy":"2024-10-16T08:45:14.024877Z","iopub.execute_input":"2024-10-16T08:45:14.025327Z","iopub.status.idle":"2024-10-16T08:45:14.554751Z","shell.execute_reply.started":"2024-10-16T08:45:14.025278Z","shell.execute_reply":"2024-10-16T08:45:14.553854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split the data into train and validation sets\nX_train, X_val, y_train, y_val, w_train, w_val = train_test_split(X, y, weights, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T08:45:24.103263Z","iopub.execute_input":"2024-10-16T08:45:24.103726Z","iopub.status.idle":"2024-10-16T08:45:26.475837Z","shell.execute_reply.started":"2024-10-16T08:45:24.103685Z","shell.execute_reply":"2024-10-16T08:45:26.474851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Scale the features (optional but can improve performance)\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_val_scaled = scaler.transform(X_val)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T08:46:46.825879Z","iopub.execute_input":"2024-10-16T08:46:46.826780Z","iopub.status.idle":"2024-10-16T08:46:50.919175Z","shell.execute_reply.started":"2024-10-16T08:46:46.826738Z","shell.execute_reply":"2024-10-16T08:46:50.918321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train a simple LightGBM model\nlgb_train = lgb.Dataset(X_train_scaled, y_train, weight=w_train)\nlgb_val = lgb.Dataset(X_val_scaled, y_val, weight=w_val, reference=lgb_train)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T08:46:54.301564Z","iopub.execute_input":"2024-10-16T08:46:54.302020Z","iopub.status.idle":"2024-10-16T08:46:54.307555Z","shell.execute_reply.started":"2024-10-16T08:46:54.301974Z","shell.execute_reply":"2024-10-16T08:46:54.306461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LightGBM parameters with GPU support and reduced complexity\nparams = {\n    'objective': 'regression',\n    'metric': 'rmse',\n    'boosting_type': 'gbdt',\n    'learning_rate': 0.05,\n    'num_leaves': 16,  # Reduced complexity\n    'feature_fraction': 0.8,\n    'bagging_fraction': 0.7,\n    'bagging_freq': 5,\n    'verbose': -1,\n    'device': 'gpu',  # Enable GPU usage\n    'gpu_platform_id': 0,\n    'gpu_device_id': 0,\n    'gpu_use_dp': False  # Single precision for faster computation\n}","metadata":{"execution":{"iopub.status.busy":"2024-10-16T08:47:22.266689Z","iopub.execute_input":"2024-10-16T08:47:22.267457Z","iopub.status.idle":"2024-10-16T08:47:22.273151Z","shell.execute_reply.started":"2024-10-16T08:47:22.267411Z","shell.execute_reply":"2024-10-16T08:47:22.272125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\nfrom lightgbm.callback import early_stopping, log_evaluation\n\n# Train the model with fewer rounds and early stopping\nprint(\"Training the LightGBM model with GPU...\")\nmodel = lgb.train(\n    params, \n    lgb_train, \n    valid_sets=[lgb_train, lgb_val], \n    num_boost_round=200, \n    callbacks=[\n        early_stopping(stopping_rounds=20),  # Early stopping callback\n        log_evaluation(50)  # Logs evaluation results every 50 rounds\n    ]\n)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T08:49:03.388614Z","iopub.execute_input":"2024-10-16T08:49:03.389049Z","iopub.status.idle":"2024-10-16T08:49:34.734681Z","shell.execute_reply.started":"2024-10-16T08:49:03.389009Z","shell.execute_reply":"2024-10-16T08:49:34.733706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make predictions on validation set\ny_val_pred = model.predict(X_val_scaled)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-16T08:50:13.086423Z","iopub.execute_input":"2024-10-16T08:50:13.087287Z","iopub.status.idle":"2024-10-16T08:50:14.725834Z","shell.execute_reply.started":"2024-10-16T08:50:13.087231Z","shell.execute_reply":"2024-10-16T08:50:14.724851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model using weighted R2 score\ndef weighted_r2_score(y_true, y_pred, weights):\n    y_mean = np.average(y_true, weights=weights)\n    ss_total = np.sum(weights * (y_true - y_mean) ** 2)\n    ss_residual = np.sum(weights * (y_true - y_pred) ** 2)\n    return 1 - (ss_residual / ss_total)\n\nr2 = weighted_r2_score(y_val, y_val_pred, w_val)\nprint(f\"Weighted R2 score on validation set: {r2:.4f}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-16T08:50:17.888751Z","iopub.execute_input":"2024-10-16T08:50:17.889151Z","iopub.status.idle":"2024-10-16T08:50:17.902918Z","shell.execute_reply.started":"2024-10-16T08:50:17.889112Z","shell.execute_reply":"2024-10-16T08:50:17.901908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(X_val_scaled)  # Get predictions\nprint(f\"Length of predictions: {len(predictions)}\")  # Check length of predictions\nprint(f\"Length of sample submission: {len(sample_submission)}\")  # Check length of submission\n","metadata":{"execution":{"iopub.status.busy":"2024-10-16T09:00:15.589038Z","iopub.execute_input":"2024-10-16T09:00:15.589519Z","iopub.status.idle":"2024-10-16T09:00:17.261473Z","shell.execute_reply.started":"2024-10-16T09:00:15.589476Z","shell.execute_reply":"2024-10-16T09:00:17.260493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Ensure your sample_submission matches the test dataset\nsample_submission['responder_6'] = predictions[:len(sample_submission)]  # Ensure it doesn't exceed the length\n# Ensure your sample_submission matches the test dataset\nsample_submission['responder_6'] = predictions[:len(sample_submission)]  # Ensure it doesn't exceed the length\n","metadata":{"execution":{"iopub.status.busy":"2024-10-16T09:01:57.136820Z","iopub.execute_input":"2024-10-16T09:01:57.137646Z","iopub.status.idle":"2024-10-16T09:01:57.143335Z","shell.execute_reply.started":"2024-10-16T09:01:57.137599Z","shell.execute_reply":"2024-10-16T09:01:57.142348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the submission file\nsample_submission.to_csv('submission.csv', index=False)\nprint(\"Submission file created.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-16T09:02:00.122525Z","iopub.execute_input":"2024-10-16T09:02:00.122925Z","iopub.status.idle":"2024-10-16T09:02:00.132047Z","shell.execute_reply.started":"2024-10-16T09:02:00.122886Z","shell.execute_reply":"2024-10-16T09:02:00.131072Z"},"trusted":true},"execution_count":null,"outputs":[]}]}