{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom matplotlib import pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\n\nimport polars as pl\nfrom sklearn.linear_model import Ridge\nimport os\nimport gc\nimport warnings\nimport kaggle_evaluation.jane_street_inference_server\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.model_selection import KFold\n\n\nimport random\n\ndef seed_everything(seed):\n    np.random.seed(seed)\n    random.seed(seed)\nseed_everything(seed=2025)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:03:48.488476Z","iopub.execute_input":"2024-12-07T11:03:48.489003Z","iopub.status.idle":"2024-12-07T11:03:49.131009Z","shell.execute_reply.started":"2024-12-07T11:03:48.488957Z","shell.execute_reply":"2024-12-07T11:03:49.129720Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path = \"/kaggle/input/jane-street-real-time-market-data-forecasting\"\nsamples = [] \n\nr = range(2)\nfor i in r:\n    file_path = f\"{path}/train.parquet/partition_id={i}/part-0.parquet\"\n    part = pd.read_parquet(file_path)\n    samples.append(part)\n    \nsample_df = pd.concat(samples, ignore_index=True)\nsample_df.round(1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:17:34.865134Z","iopub.execute_input":"2024-12-07T10:17:34.865797Z","iopub.status.idle":"2024-12-07T10:17:47.402179Z","shell.execute_reply.started":"2024-12-07T10:17:34.865751Z","shell.execute_reply":"2024-12-07T10:17:47.400901Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train =sample_df\ntrain['N']=train.index.values \ntrain['id']=train.index.values \n\ngridColor = 'lightgrey'\nxx= sample_df[(sample_df.symbol_id==1)] ['id']\nyy=sample_df[ (sample_df.symbol_id==1)]['responder_6']\n\nplt.figure(figsize=(16, 5))\nplt.plot(xx,yy, color = 'black', linewidth =0.05)\nplt.suptitle('Responder_6', weight='bold', fontsize=16)\nplt.xlabel(\"Time\", fontsize=12)\nplt.ylabel(\"Returns\", fontsize=12)\nplt.grid(color = gridColor , linewidth=0.8)\nplt.axhline(0, color='red', linestyle='-', linewidth=1.2)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:17:51.837727Z","iopub.execute_input":"2024-12-07T10:17:51.838178Z","iopub.status.idle":"2024-12-07T10:17:52.848903Z","shell.execute_reply.started":"2024-12-07T10:17:51.838140Z","shell.execute_reply":"2024-12-07T10:17:52.847715Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(14, 4))\nplt.plot(xx,yy.cumsum(), color = 'black', linewidth =0.6)\nplt.suptitle('Cumulative responder_6', weight='bold', fontsize=16)\nplt.xlabel(\"Time\", fontsize=12)\nplt.ylabel(\"Cumulative res\", fontsize=12)\nplt.yticks(np.arange(-500,1000,250))\n#plt.xticks(np.arange(0,170,10))\nplt.grid(color = gridColor)\n#plt.grid(color = 'lightblue')\nplt.axhline(0, color='red', linestyle='-', linewidth=0.7)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:17:56.164078Z","iopub.execute_input":"2024-12-07T10:17:56.165324Z","iopub.status.idle":"2024-12-07T10:17:56.438849Z","shell.execute_reply.started":"2024-12-07T10:17:56.165269Z","shell.execute_reply":"2024-12-07T10:17:56.437627Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(18, 7))\npredictor_cols = [col for col in sample_df.columns if 'responder' in col]\nfor i in predictor_cols: \n    if i == 'responder_6': \n        c='red'\n        lw=2.5\n        plt.plot((sample_df[sample_df.symbol_id == 0].groupby(['date_id'])[i].mean()).cumsum(), linewidth = lw, color = c)\n    else: \n        lw=1\n        plt.plot((sample_df[sample_df.symbol_id == 0].groupby(['date_id'])[i].mean()).cumsum(), linewidth = lw)\n\nplt.xlabel('Trade days')\nplt.ylabel('Cumulative response')\nplt.title('Response time series over trade days  \\n Responder 6 (red) and other responders', weight='bold')\nplt.grid(visible=True, color = gridColor, linewidth = 0.7)\nplt.axhline(0, color='blue', linestyle='-', linewidth=1)\nplt.legend(predictor_cols)\nsns.despine()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:17:59.958590Z","iopub.execute_input":"2024-12-07T10:17:59.959007Z","iopub.status.idle":"2024-12-07T10:18:02.425598Z","shell.execute_reply.started":"2024-12-07T10:17:59.958973Z","shell.execute_reply":"2024-12-07T10:18:02.423801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features = pd.read_csv(f\"{path}/features.csv\")\nfeatures","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:11:16.544597Z","iopub.execute_input":"2024-12-07T10:11:16.545079Z","iopub.status.idle":"2024-12-07T10:11:16.585884Z","shell.execute_reply.started":"2024-12-07T10:11:16.545025Z","shell.execute_reply":"2024-12-07T10:11:16.584768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"responders = pd.read_csv(f\"{path}/responders.csv\")\nresponders","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:11:16.587334Z","iopub.execute_input":"2024-12-07T10:11:16.587764Z","iopub.status.idle":"2024-12-07T10:11:16.606218Z","shell.execute_reply.started":"2024-12-07T10:11:16.587716Z","shell.execute_reply":"2024-12-07T10:11:16.604847Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df['weight'].describe().round(1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:11:16.607633Z","iopub.execute_input":"2024-12-07T10:11:16.608145Z","iopub.status.idle":"2024-12-07T10:11:16.745368Z","shell.execute_reply.started":"2024-12-07T10:11:16.608103Z","shell.execute_reply":"2024-12-07T10:11:16.744113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(8,3))\nplt.hist(sample_df['weight'], bins=30, color='grey', edgecolor = 'white',density=True )\nplt.title('Distribution of weights')\nplt.grid(color = 'lightgrey', linewidth=0.5)\nplt.axvline(1.7, color='red', linestyle='-', linewidth=0.7)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:11:16.746428Z","iopub.execute_input":"2024-12-07T10:11:16.746750Z","iopub.status.idle":"2024-12-07T10:11:17.037348Z","shell.execute_reply.started":"2024-12-07T10:11:16.746720Z","shell.execute_reply":"2024-12-07T10:11:17.036186Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport polars as pl\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import GridSearchCV, KFold\nfrom sklearn.metrics import make_scorer\nimport gc\n\ndef custom_metric(y_true,y_pred,weight):\n    weighted_r2=1-(np.sum(weight*(y_true-y_pred)**2)/np.sum(weight*y_true**2))\n    return weighted_r2\n\nprint(\"< read and process data in chunks >\")\nchunk_size = 1\ntrain_X_list, train_y_list, weights_list = [], [], []\n\nfor start_partition in range(6, 10, chunk_size):\n    for i in range(start_partition, min(start_partition + chunk_size, 10)):\n\n        train = pl.read_parquet(f\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={i}/part-0.parquet\")\n        train = train.to_pandas().sample(frac=0.97, random_state=2025)\n        \n        weights_list.extend(train['weight'].values)\n        train_y_list.extend(train['responder_6'].values)\n        \n        train.drop(['weight', 'responder_6'], axis=1, inplace=True)\n        cols = [f'feature_0{i}' if i < 10 else f'feature_{i}' for i in range(79)]\n        train_X_list.append(train[cols].fillna(3).values)\n        \n        del train\n        gc.collect()\n\ntrain_X = np.vstack(train_X_list)\ntrain_y = np.array(train_y_list)\nweights = np.array(weights_list)\n\ndel train_X_list, train_y_list, weights_list\ngc.collect()\n\nprint(f\"train_X.shape: {train_X.shape}, train_y.shape: {train_y.shape}\")\n\nprint(\"< train test split >\")\nsplit = 40000\ntrain_X, test_X = train_X[:-split], train_X[-split:]\ntrain_y, test_y = train_y[:-split], train_y[-split:]\ntrain_weight, test_weight = weights[:-split], weights[-split:]\nprint(f\"train_X.shape: {train_X.shape}, test_X.shape: {test_X.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:57:17.365423Z","iopub.execute_input":"2024-12-07T10:57:17.365799Z","iopub.status.idle":"2024-12-07T10:59:23.118488Z","shell.execute_reply.started":"2024-12-07T10:57:17.365760Z","shell.execute_reply":"2024-12-07T10:59:23.117225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"< hyperparameter tuning >\")\nscorer = make_scorer(lambda y_true, y_pred, weights: custom_metric(y_true, y_pred, weights), greater_is_better=True)\n\nparam_grid = {'alpha': [0.01, 0.1, 1, 10, 100]}\nridge = Ridge()\nsample_indices = np.random.choice(len(train_X), size=50000, replace=False)\ngrid_search = GridSearchCV(ridge, param_grid, scoring=scorer, cv=3, n_jobs=-1, verbose=2)\ngrid_search.fit(train_X[sample_indices], train_y[sample_indices])\nbest_alpha = grid_search.best_params_['alpha']\nprint(f\"Best alpha: {best_alpha}\")\n\n\nprint(\"< cross-validation >\")\nkf = KFold(n_splits=5, shuffle=True, random_state=2025)\ncv_scores = []\n\nfor fold, (train_idx, val_idx) in enumerate(kf.split(train_X)):\n    fold_train_X, fold_val_X = train_X[train_idx], train_X[val_idx]\n    fold_train_y, fold_val_y = train_y[train_idx], train_y[val_idx]\n    fold_train_weight, fold_val_weight = train_weight[train_idx], train_weight[val_idx]\n\n    model = Ridge(alpha=best_alpha)\n    model.fit(fold_train_X, fold_train_y)\n\n    val_pred = model.predict(fold_val_X)\n    score = custom_metric(fold_val_y, val_pred, fold_val_weight)\n    cv_scores.append(score)\n    print(f\"Fold {fold + 1}: Weighted R2 = {score}\")\n\nmean_cv_score = np.mean(cv_scores)\nprint(f\"Mean Weighted R2 across folds: {mean_cv_score}\")\n\nprint(\"< fit and predict >\")\nmodel = Ridge(alpha=best_alpha)\nmodel.fit(train_X, train_y)\ntrain_pred = model.predict(train_X)\ntest_pred = model.predict(test_X)\nprint(f\"train weighted_r2: {custom_metric(train_y, train_pred, weight=train_weight)}\")\nprint(f\"test weighted_r2: {custom_metric(test_y, test_pred, weight=test_weight)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:59:28.262871Z","iopub.execute_input":"2024-12-07T10:59:28.263410Z","iopub.status.idle":"2024-12-07T11:01:33.389021Z","shell.execute_reply.started":"2024-12-07T10:59:28.263371Z","shell.execute_reply":"2024-12-07T11:01:33.387440Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict(test,lags):\n    cols=[f'feature_0{i}' if i<10 else f'feature_{i}' for i in range(79)]\n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n    test=test.to_pandas()[cols].fillna(3)\n    test_preds=model.predict(test.values)\n    predictions = predictions.with_columns(pl.Series('responder_6', test_preds.ravel()))\n    return predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:03:30.810549Z","iopub.execute_input":"2024-12-07T11:03:30.811681Z","iopub.status.idle":"2024-12-07T11:03:30.819803Z","shell.execute_reply.started":"2024-12-07T11:03:30.811633Z","shell.execute_reply":"2024-12-07T11:03:30.818285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:04:30.536579Z","iopub.execute_input":"2024-12-07T11:04:30.537048Z","iopub.status.idle":"2024-12-07T11:04:30.576961Z","shell.execute_reply.started":"2024-12-07T11:04:30.537008Z","shell.execute_reply":"2024-12-07T11:04:30.575217Z"}},"outputs":[],"execution_count":null}]}