{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":9801075,"sourceType":"datasetVersion","datasetId":6006872},{"sourceId":9806342,"sourceType":"datasetVersion","datasetId":6010899},{"sourceId":203900450,"sourceType":"kernelVersion"},{"sourceId":171905,"sourceType":"modelInstanceVersion","modelInstanceId":146319,"modelId":168862}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":80.344101,"end_time":"2024-10-26T03:27:42.952247","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-10-26T03:26:22.608146","version":"2.6.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"252dd2de87de42f4becd877fbbafc26b":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_ff55584ae0ab4f39ae471f314eaad988","placeholder":"​","style":"IPY_MODEL_b8d338c473aa4dc3ba25f37a997a9037","value":" 1/1 [00:00&lt;00:00, 33.79it/s]"}},"48a2731fb59b4ce8ace5b53d6f0e3337":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_a8dbc79c7c5a48318466a102c55801bb","placeholder":"​","style":"IPY_MODEL_ebd06aaaa7024a1684ba1b9fe89358cf","value":"100%"}},"56da652f3eeb42aca986e6fd815629dc":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"5bb4e0df01c44716afebab25aafe9f5f":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"ProgressView","bar_style":"success","description":"","description_tooltip":null,"layout":"IPY_MODEL_f4e41234b0cc4f3e9313d971f5aadc9f","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_6408698650f74dd699bcc914b850e396","value":1}},"6408698650f74dd699bcc914b850e396":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"9cb493ec04fc4ed391f5ac28cc84500e":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_48a2731fb59b4ce8ace5b53d6f0e3337","IPY_MODEL_5bb4e0df01c44716afebab25aafe9f5f","IPY_MODEL_252dd2de87de42f4becd877fbbafc26b"],"layout":"IPY_MODEL_56da652f3eeb42aca986e6fd815629dc"}},"a8dbc79c7c5a48318466a102c55801bb":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"b8d338c473aa4dc3ba25f37a997a9037":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"ebd06aaaa7024a1684ba1b9fe89358cf":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"f4e41234b0cc4f3e9313d971f5aadc9f":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"ff55584ae0ab4f39ae471f314eaad988":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## **EDA**","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv, pd.read_parquet )\nimport polars as pl\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom matplotlib import pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\n\nimport os, gc\nfrom tqdm.auto import tqdm\nimport pickle # module to serialize and deserialize objects\nimport re # for Regular expression operations \n\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom tensorflow.keras.optimizers import Adam\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data  import Dataset, DataLoader\nfrom pytorch_lightning import (LightningDataModule, LightningModule, Trainer)\nfrom pytorch_lightning.callbacks import EarlyStopping, ModelCheckpoint, Timer\n\nfrom sklearn.metrics import r2_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import VotingRegressor\n\nimport lightgbm as lgb\nfrom lightgbm import LGBMRegressor\n\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\n\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nimport kaggle_evaluation.jane_street_inference_server","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:47:21.008897Z","iopub.execute_input":"2024-12-01T06:47:21.009768Z","iopub.status.idle":"2024-12-01T06:47:39.047747Z","shell.execute_reply.started":"2024-12-01T06:47:21.009731Z","shell.execute_reply":"2024-12-01T06:47:39.047006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gridColor = 'lightgrey'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:47:39.049284Z","iopub.execute_input":"2024-12-01T06:47:39.049956Z","iopub.status.idle":"2024-12-01T06:47:39.053984Z","shell.execute_reply.started":"2024-12-01T06:47:39.049926Z","shell.execute_reply":"2024-12-01T06:47:39.053154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\npath = \"/kaggle/input/jane-street-real-time-market-data-forecasting\"\nsamples = [] \n\n# Load a data from each file:\nr = [8, 9]\nfor i in r:\n    file_path = f\"{path}/train.parquet/partition_id={i}/part-0.parquet\"\n    part = pd.read_parquet(file_path)\n    samples.append(part)\n    \nsample_df = pd.concat(samples, ignore_index=True) # Concatenate all samples into one DataFrame if needed\n\nsample_df.round(1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:47:39.055434Z","iopub.execute_input":"2024-12-01T06:47:39.055950Z","iopub.status.idle":"2024-12-01T06:48:02.326022Z","shell.execute_reply.started":"2024-12-01T06:47:39.055909Z","shell.execute_reply":"2024-12-01T06:48:02.325119Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(sample_df.shape)\nprint(sample_df.date_id.unique().size)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:35:22.884913Z","iopub.execute_input":"2024-12-01T06:35:22.885834Z","iopub.status.idle":"2024-12-01T06:35:22.943473Z","shell.execute_reply.started":"2024-12-01T06:35:22.885790Z","shell.execute_reply":"2024-12-01T06:35:22.942713Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We can see that two files (8 and 9) has a total of 12414600 rows. \nI have used pandas to load it and it took almost 9 sec. Th ere are a total of 339 days  (about one years of trading data).  ","metadata":{}},{"cell_type":"markdown","source":"Time Series Analysis & EDA","metadata":{}},{"cell_type":"markdown","source":"Now let's compare this responder (6) with other responders","metadata":{}},{"cell_type":"code","source":"# for symbol_id == 0\nplt.figure(figsize=(18, 7))\npredictor_cols = [col for col in sample_df.columns if 'responder' in col]\nfor i in predictor_cols: \n    if i == 'responder_6': \n        c='red'\n        lw=2.5\n        plt.plot((sample_df[sample_df.symbol_id == 0].groupby(['date_id'])[i].mean()).cumsum(), linewidth = lw, color = c)\n    else: \n        lw=1\n        plt.plot((sample_df[sample_df.symbol_id == 0].groupby(['date_id'])[i].mean()).cumsum(), linewidth = lw)\n\nplt.xlabel('Trade days')\nplt.ylabel('Cumulative response')\nplt.title('Response time series over trade days  \\n Responder 6 (red) and other responders', weight='bold')\nplt.grid(visible=True, color = gridColor, linewidth = 0.7)\nplt.axhline(0, color='blue', linestyle='-', linewidth=1)\nplt.legend(predictor_cols)\nsns.despine()\n#plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:35:24.267279Z","iopub.execute_input":"2024-12-01T06:35:24.267551Z","iopub.status.idle":"2024-12-01T06:35:27.276915Z","shell.execute_reply.started":"2024-12-01T06:35:24.267525Z","shell.execute_reply":"2024-12-01T06:35:27.276006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Correlation between responder 6 to responder\ndef apply_lag_on_non_responder_6(df, n_lag):\n    df['responder_7'] = df.responder_7.shift(n_lag)\n    df['responder_8'] = df.responder_8.shift(n_lag)\n    return df.reset_index(level=0, drop=True)\n\nresults = []\nfor n_lag in range(1, 20):\n    result = (\n        sample_df[['symbol_id', 'date_id', 'time_id', 'responder_6', 'responder_7', 'responder_8']]\n        .groupby('symbol_id').apply(lambda df: apply_lag_on_non_responder_6(df, n_lag)).groupby(level=0)[['responder_6', 'responder_7', 'responder_8']].corr()\n    ).groupby(level=1).mean()\n    results.append({\n        'n_lag': n_lag,\n        'corr_6_7': result.iloc[1, 0],\n        'corr_6_8': result.iloc[2, 0],\n    })\nresults = pd.DataFrame(results)    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:35:27.278403Z","iopub.execute_input":"2024-12-01T06:35:27.279116Z","iopub.status.idle":"2024-12-01T06:36:19.044619Z","shell.execute_reply.started":"2024-12-01T06:35:27.279052Z","shell.execute_reply":"2024-12-01T06:36:19.043780Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results.set_index('n_lag').plot(kind='bar')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:36:19.045677Z","iopub.execute_input":"2024-12-01T06:36:19.045941Z","iopub.status.idle":"2024-12-01T06:36:19.433939Z","shell.execute_reply.started":"2024-12-01T06:36:19.045914Z","shell.execute_reply":"2024-12-01T06:36:19.433021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Correlation between responders in daily close\nsns.heatmap(\n    sample_df.groupby(['symbol_id', 'date_id']).last().filter(regex='responder_[\\d]+').groupby(level=0).corr().groupby(level=1).mean()\n)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:36:19.438329Z","iopub.execute_input":"2024-12-01T06:36:19.438712Z","iopub.status.idle":"2024-12-01T06:36:22.767793Z","shell.execute_reply.started":"2024-12-01T06:36:19.438681Z","shell.execute_reply":"2024-12-01T06:36:22.766977Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- We can see that `resp6` (red) most closely follows `resp0` and `resp3`\n\nLet's build a correlation matrix and see it numerically.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(6, 6))\nresponders = pd.read_csv(f\"{path}/responders.csv\")\nmatrix = responders[[ f\"tag_{no}\" for no in range(0,5,1) ] ].T.corr()\nsns.heatmap(matrix, square=True, cmap=\"coolwarm\", alpha =0.9, vmin=-1, vmax=1, center= 0, linewidths=0.5, \n            linecolor='white', annot=True, fmt='.2f')\nplt.xlabel(\"Responder_0 - Responder_8\")\nplt.ylabel(\"Responder_0 - Responder_8\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:36:22.769087Z","iopub.execute_input":"2024-12-01T06:36:22.769459Z","iopub.status.idle":"2024-12-01T06:36:23.140886Z","shell.execute_reply.started":"2024-12-01T06:36:22.769417Z","shell.execute_reply":"2024-12-01T06:36:23.139624Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Let us take a look at the returns and cumulative daily returns, and disribution of returns for all responders","metadata":{}},{"cell_type":"code","source":"df_train=sample_df\ns_id = 0                        # Change params to take a look at other symbols\nres_columns = [col for col in df_train.columns if re.match(\"responder_\", col)]\nrow = 9\nj = 0\n\nfig, axs = plt.subplots(figsize=(18, 4*row))\nfor i in range(1, 3 * len(res_columns) + 1, 3):\n    xx= sample_df[(sample_df.symbol_id==s_id)] ['N']\n    yy=sample_df[ (sample_df.symbol_id==s_id)][f'responder_{j}']\n    c='black'\n    if j == 6: c='red'\n        \n    ax1 = plt.subplot(9, 3, i)\n    ax1.plot(   xx,yy.cumsum()   , color = c, linewidth =0.8 )\n    plt.axhline(0, color='blue', linestyle='-', linewidth=0.9)\n    plt.grid(color =gridColor )\n    \n    ax2 = plt.subplot(9, 3, i+1)\n    #by_date = df_symbolX.groupby([\"date_id\"])\n    ax2.plot(xx,yy   , color = c, linewidth =0.05)\n    plt.axhline(0, color='blue', linestyle='-', linewidth=1.2)\n    ax2.set_title(f\"responder_{j}\", fontsize = 14)\n    plt.grid(color = gridColor)\n    \n    ax3 = plt.subplot(9, 3, i+2)\n    b=1000\n    ax3.hist(yy, bins=b, color = c,density=True, histtype=\"step\" )\n    ax3.hist(yy, bins=b, color = 'lightgrey',density=True)\n    plt.grid(color = gridColor)\n    ax3.set_ylim([0, 3.5])\n    ax3.set_xlim([-2.5, 2.5])\n    \n    j = j + 1\n    \nfig.patch.set_linewidth(3)\nfig.patch.set_edgecolor('#000000')\nfig.patch.set_facecolor('#eeeeee') \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:36:23.142275Z","iopub.execute_input":"2024-12-01T06:36:23.142661Z","iopub.status.idle":"2024-12-01T06:36:44.337050Z","shell.execute_reply.started":"2024-12-01T06:36:23.142612Z","shell.execute_reply":"2024-12-01T06:36:44.336217Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We can see that responders have different behavior and distributions.\n\nLet us now study the behavior of `responder 6`  for different `symbol_id`","metadata":{}},{"cell_type":"code","source":"res_columns = [col for col in df_train.columns if re.match(\"responder_\", col)]\nrow=10\nfig, axs = plt.subplots(figsize=(18, 5*row))\nb=300\nj = 0\nfor i in range(1, 3 * row + 1, 3):\n    xx= sample_df[(sample_df.symbol_id==j)] ['N']\n    yy= sample_df[(sample_df.symbol_id==j)]['responder_6']\n    c='black'\n        \n    ax1 = plt.subplot(row, 3, i)\n    ax1.plot(   xx,yy.cumsum()   , color = c, linewidth =0.8 )\n    plt.axhline(0, color='red', linestyle='-', linewidth=0.7)\n    plt.grid(color = gridColor)\n    plt.xlabel('Time')\n    \n    ax2 = plt.subplot(row, 3, i+1)\n    ax2.plot(xx,yy   , color = c, linewidth =0.05)\n    plt.axhline(0, color='red', linestyle='-', linewidth=0.7)\n    ax2.set_title(f\"symbol_id={j}\", fontsize = '14')\n    plt.grid(color = gridColor)\n    plt.xlabel('Time')\n    \n    ax3 = plt.subplot(row, 3, i+2)\n    ax3.hist(yy, bins=b, color = c, density=True, histtype=\"step\" )\n    ax3.hist(yy, bins=b, color = 'lightgrey',density=True)\n    plt.grid(color = gridColor)\n    ax3.set_xlim([-2.5, 2.5])\n    ax3.set_ylim([0, 1.5])\n    plt.xlabel('Time')\n    \n    j = j + 1\n    \nfig.patch.set_linewidth(3)\nfig.patch.set_edgecolor('#000000')\nfig.patch.set_facecolor('#eeeeee') \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:36:44.338387Z","iopub.execute_input":"2024-12-01T06:36:44.338726Z","iopub.status.idle":"2024-12-01T06:36:59.697828Z","shell.execute_reply.started":"2024-12-01T06:36:44.338690Z","shell.execute_reply":"2024-12-01T06:36:59.696826Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- We see that the behavior and distribution of one `responder 6` is  different for different `symbol_id`\n\nNow let's study the data in more detail and then continue diving into time series analysis\n","metadata":{}},{"cell_type":"markdown","source":"## Files and variables overview","metadata":{}},{"cell_type":"markdown","source":"### Features.csv\nfeatures.csv - metadata pertaining to the anonymized features\n\n#### Features have many missing values.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(20, 3))    # Plot missing values\nplt.bar(x=sample_df.isna().sum().index, height=sample_df.isna().sum().values, color=\"red\", label='missing')   # analog: using missingno\nplt.xticks(rotation=90)\nplt.title(f'Missing values over the {len(df_train)} samples which have a target')\nplt.grid()\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:36:59.699546Z","iopub.execute_input":"2024-12-01T06:36:59.699894Z","iopub.status.idle":"2024-12-01T06:37:02.934808Z","shell.execute_reply.started":"2024-12-01T06:36:59.699859Z","shell.execute_reply":"2024-12-01T06:37:02.933860Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df.isna().sum().sort_values(ascending=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:37:02.936285Z","iopub.execute_input":"2024-12-01T06:37:02.937025Z","iopub.status.idle":"2024-12-01T06:37:04.147551Z","shell.execute_reply.started":"2024-12-01T06:37:02.936973Z","shell.execute_reply":"2024-12-01T06:37:04.146629Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- Some columns are not very useful in our sample (either Null or show the partition number).","metadata":{}},{"cell_type":"markdown","source":"#### Structure of features:","metadata":{}},{"cell_type":"code","source":"features = pd.read_csv(f\"{path}/features.csv\")\nfeatures","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:37:04.148891Z","iopub.execute_input":"2024-12-01T06:37:04.149646Z","iopub.status.idle":"2024-12-01T06:37:04.173827Z","shell.execute_reply.started":"2024-12-01T06:37:04.149591Z","shell.execute_reply":"2024-12-01T06:37:04.172978Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Tags visualizing:","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(18, 6))\nplt.imshow(features.iloc[:, 1:].T.values, cmap=\"gray_r\")\nplt.xlabel(\"feature_00 - feature_78\")\nplt.ylabel(\"tag_0 - tag_16\")\nplt.yticks(np.arange(17))\nplt.xticks(np.arange(79))\nplt.grid(color = 'lightgrey')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:37:04.174835Z","iopub.execute_input":"2024-12-01T06:37:04.175104Z","iopub.status.idle":"2024-12-01T06:37:04.752580Z","shell.execute_reply.started":"2024-12-01T06:37:04.175051Z","shell.execute_reply":"2024-12-01T06:37:04.751500Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Correlation matrix between feature_XX and feature_YY","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(11, 11))\nmatrix = sample_df.filter(regex='feature.*').corr()\nsns.heatmap(matrix, square=True, cmap=\"coolwarm\", alpha =0.9, vmin=-1, vmax=1, center= 0, linewidths=0.5, linecolor='white')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:37:04.753685Z","iopub.execute_input":"2024-12-01T06:37:04.753932Z","iopub.status.idle":"2024-12-01T06:40:12.923953Z","shell.execute_reply.started":"2024-12-01T06:37:04.753907Z","shell.execute_reply":"2024-12-01T06:40:12.923120Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pairs = []\nfor i in range(len(matrix)):\n    for j in range(len(matrix)):\n        if matrix.iloc[i, j]>0.9 and i != j and j >i :\n            print(i, j, matrix.iloc[i,j])\n            pairs.append([i, j, matrix.iloc[i, j]])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:40:12.925333Z","iopub.execute_input":"2024-12-01T06:40:12.925713Z","iopub.status.idle":"2024-12-01T06:40:13.042839Z","shell.execute_reply.started":"2024-12-01T06:40:12.925672Z","shell.execute_reply":"2024-12-01T06:40:13.042010Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_df = sample_df[\n    (sample_df.symbol_id.isin(sample_df.symbol_id.unique()[::4])) &\n    (sample_df.time_id.isin(sample_df.time_id.unique()[::4]))\n]\nprint(plot_df.shape)\n\nfor pair in pairs:\n    sns.scatterplot(plot_df, x=f'feature_{pair[0]:02}', y=f'feature_{pair[1]:02}')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:40:13.043845Z","iopub.execute_input":"2024-12-01T06:40:13.044220Z","iopub.status.idle":"2024-12-01T06:40:43.363352Z","shell.execute_reply.started":"2024-12-01T06:40:13.044180Z","shell.execute_reply":"2024-12-01T06:40:43.362460Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 73 74, 75,76, 77,78 are highly correlated and belong to the same groups\n# Will use average to merge the two features\nmerge_groups = [[73, 74], [75, 76], [77,78]]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:40:43.364732Z","iopub.execute_input":"2024-12-01T06:40:43.365149Z","iopub.status.idle":"2024-12-01T06:40:43.369733Z","shell.execute_reply.started":"2024-12-01T06:40:43.365105Z","shell.execute_reply":"2024-12-01T06:40:43.368764Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Response 21-","metadata":{}},{"cell_type":"markdown","source":"### Responders.csv\nresponders.csv - metadata pertaining to the anonymized responders\n#### Structure of responders:","metadata":{}},{"cell_type":"code","source":"responders = pd.read_csv(f\"{path}/responders.csv\")\nresponders","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:40:43.370653Z","iopub.execute_input":"2024-12-01T06:40:43.370903Z","iopub.status.idle":"2024-12-01T06:40:43.389572Z","shell.execute_reply.started":"2024-12-01T06:40:43.370859Z","shell.execute_reply":"2024-12-01T06:40:43.388849Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Weights\n#### Basic stats:","metadata":{}},{"cell_type":"code","source":"sample_df['weight'].describe().round(1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:40:43.390640Z","iopub.execute_input":"2024-12-01T06:40:43.390985Z","iopub.status.idle":"2024-12-01T06:40:43.757252Z","shell.execute_reply.started":"2024-12-01T06:40:43.390947Z","shell.execute_reply":"2024-12-01T06:40:43.756261Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"> ","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8,3))\nplt.hist(sample_df['weight'], bins=30, color='grey', edgecolor = 'white',density=True )\nplt.title('Distribution of weights')\nplt.grid(color = 'lightgrey', linewidth=0.5)\nplt.axvline(1.7, color='red', linestyle='-', linewidth=0.7)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:40:43.758444Z","iopub.execute_input":"2024-12-01T06:40:43.758830Z","iopub.status.idle":"2024-12-01T06:40:44.140978Z","shell.execute_reply.started":"2024-12-01T06:40:43.758789Z","shell.execute_reply":"2024-12-01T06:40:44.140128Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Sample submission.csv\nsample_submission.csv - This file illustrates the format of the predictions your model should make.","metadata":{}},{"cell_type":"code","source":"sub = pd.read_csv(f\"{path}/sample_submission.csv\")\nprint( f\"shape = {sub.shape}\" )\nsub.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:40:44.142087Z","iopub.execute_input":"2024-12-01T06:40:44.142396Z","iopub.status.idle":"2024-12-01T06:40:44.156042Z","shell.execute_reply.started":"2024-12-01T06:40:44.142367Z","shell.execute_reply":"2024-12-01T06:40:44.155253Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Train.parquet\n\n- **train.parquet** - The training set, contains historical data and returns. For convenience, the training set has been partitioned into ten parts.\n  - `date_id` and `time_id` - Integer values that are ordinally sorted, providing a chronological structure to the data, although the actual time intervals between `time_id` values may vary.\n  - `symbol_id` - Identifies a unique financial instrument.\n  - `weight` - The weighting used for calculating the scoring function.\n  - `feature_{00...78}` - Anonymized market data.\n  - `responder_{0...8}` - Anonymized responders clipped between -5 and 5. The `responder_6` field is what you are trying to predict.\n  \n  \nEach row in the `{train/test}.parquet` dataset corresponds to a unique combination of a symbol (identified by `symbol_id`) and a timestamp (represented by `date_id` and `time_id`). You will be provided with multiple responders, with `responder_6` being the only responder used for scoring. The date_id column is an integer which represents the day of the event, while `time_id` represents a time ordering. It's important to note that the real time differences between each time_id are not guaranteed to be consistent.\n\n- The `symbol_id` column contains encrypted identifiers. Each `symbol_id` is not guaranteed to appear in all `time_id` and `date_id` combinations.\n- Additionally, new `symbol_id` values **may appear in future** test sets.est sets.","metadata":{}},{"cell_type":"markdown","source":"## Responders: analysis, statistics and distributions","metadata":{}},{"cell_type":"code","source":"col =[]\nfor i in range(9):\n    col.append(f\"responder_{i}\") \n\nsample_df[col].describe().round(1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:40:44.156941Z","iopub.execute_input":"2024-12-01T06:40:44.157212Z","iopub.status.idle":"2024-12-01T06:40:48.053117Z","shell.execute_reply.started":"2024-12-01T06:40:44.157186Z","shell.execute_reply":"2024-12-01T06:40:48.052117Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Interesting fact:\n- The values ​​of all variables are strictly within the range of `[-5, 5]`","metadata":{}}]}