{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport polars as pl\nimport polars.selectors as cs\nfrom glob import glob\nfrom pprint import pprint\nfrom itertools import combinations\nfrom tqdm.contrib.concurrent import thread_map\nfrom tqdm.asyncio import tqdm as async_tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:44:59.883522Z","iopub.execute_input":"2024-12-07T10:44:59.883951Z","iopub.status.idle":"2024-12-07T10:45:00.371355Z","shell.execute_reply.started":"2024-12-07T10:44:59.883915Z","shell.execute_reply":"2024-12-07T10:45:00.369929Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style='font: 300 20px/22px \"Lato\", \"Open Sans\", \"Helvetica Neue\", Helvetica, Arial, sans-serif'>\n<p>Here we perform some <strong>resource-intensive</strong> and <strong>time-consuming</strong> operations. As a result, we will prepare several tables that we will transfer to the <strong><a href=\"https://www.kaggle.com/code/pib73nl/janestreet-rtmdf\">EDA notebook</a></strong>, where the visualization will be performed.</p>\n<p>Running this notebook on a simple Kaggle instance takes just under 20 minutes.</p>\n</div>","metadata":{}},{"cell_type":"code","source":"MAIN_PATH = '/kaggle/input/jane-street-real-time-market-data-forecasting'\nTRAIN_PATH = MAIN_PATH + '/train.parquet'\nFEATURES_NUMBER = 79\nRESPONDERS_NUMBER = 9","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:45:00.373417Z","iopub.execute_input":"2024-12-07T10:45:00.373794Z","iopub.status.idle":"2024-12-07T10:45:00.379008Z","shell.execute_reply.started":"2024-12-07T10:45:00.373748Z","shell.execute_reply":"2024-12-07T10:45:00.377785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_files = sorted([path for path in glob(f'{TRAIN_PATH}/*/*.parquet')])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:45:00.380643Z","iopub.execute_input":"2024-12-07T10:45:00.381138Z","iopub.status.idle":"2024-12-07T10:45:00.443125Z","shell.execute_reply.started":"2024-12-07T10:45:00.381092Z","shell.execute_reply":"2024-12-07T10:45:00.441942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def lazy_df_prep(parts: list) -> pl.LazyFrame:\n    \"\"\"\n    Prepare a LazyFrame from one or more parts (files)\n    \"\"\"\n    \n    list_for_concat = [pl.scan_parquet(x) for x in parts]\n    df = pl.concat(list_for_concat)\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:45:00.445234Z","iopub.execute_input":"2024-12-07T10:45:00.445585Z","iopub.status.idle":"2024-12-07T10:45:00.452183Z","shell.execute_reply.started":"2024-12-07T10:45:00.445550Z","shell.execute_reply":"2024-12-07T10:45:00.450768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_feat_n_resp = lazy_df_prep(train_files).select(pl.all().exclude(['date_id', 'time_id', 'symbol_id', 'weight']))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:45:00.453525Z","iopub.execute_input":"2024-12-07T10:45:00.454131Z","iopub.status.idle":"2024-12-07T10:45:00.498183Z","shell.execute_reply.started":"2024-12-07T10:45:00.454085Z","shell.execute_reply":"2024-12-07T10:45:00.497122Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Description","metadata":{}},{"cell_type":"markdown","source":"Calculating df.discribe() over all data fails with OOM-error. So calculate for each individual statistic and then put them together. It takes about 10 minutes","metadata":{}},{"cell_type":"code","source":"def my_describe(df):\n    \n    statistics = {'mean': df.mean(),\n                  'std':  df.std(),\n                  'min':  df.min(), \n                  '5%':   df.quantile(0.05),\n                  '25%':  df.quantile(0.25),\n                  '50%':  df.quantile(0.5),\n                  '75%':  df.quantile(0.75),\n                  '95%':  df.quantile(0.95),\n                  'max':  df.max()\n                 }\n    df_list = []\n    \n    for stat_name, stat_func in statistics.items():\n        df_list.append(stat_func\n                       .with_columns(statistic = pl.lit(stat_name))\n                       .cast({cs.numeric(): pl.Float64})\n                      )\n\n    return pl.concat(df_list)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:45:00.499465Z","iopub.execute_input":"2024-12-07T10:45:00.499947Z","iopub.status.idle":"2024-12-07T10:45:00.506941Z","shell.execute_reply.started":"2024-12-07T10:45:00.499883Z","shell.execute_reply":"2024-12-07T10:45:00.505783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\ndf_desc = my_describe(df_feat_n_resp)\ndf_desc = df_desc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:45:00.508368Z","iopub.execute_input":"2024-12-07T10:45:00.508686Z","iopub.status.idle":"2024-12-07T10:55:05.224539Z","shell.execute_reply.started":"2024-12-07T10:45:00.508646Z","shell.execute_reply":"2024-12-07T10:55:05.223299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_desc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:55:05.226095Z","iopub.execute_input":"2024-12-07T10:55:05.226543Z","iopub.status.idle":"2024-12-07T10:55:05.263112Z","shell.execute_reply.started":"2024-12-07T10:55:05.226496Z","shell.execute_reply":"2024-12-07T10:55:05.261763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# save the data to disk\ndf_desc.write_parquet('discribe.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:55:05.264860Z","iopub.execute_input":"2024-12-07T10:55:05.265297Z","iopub.status.idle":"2024-12-07T10:55:05.337856Z","shell.execute_reply.started":"2024-12-07T10:55:05.265250Z","shell.execute_reply":"2024-12-07T10:55:05.336914Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Correlation","metadata":{}},{"cell_type":"markdown","source":"Calculating correlation over all data fails with OOM-error. Therefore, we calculate the correlations in pairs and collect them into an array.","metadata":{}},{"cell_type":"code","source":"%%time\n# performing a collection in a cycle takes much longer, so we collect the entire dataframe at once\ndf = df_feat_n_resp.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:55:05.340734Z","iopub.execute_input":"2024-12-07T10:55:05.341102Z","iopub.status.idle":"2024-12-07T10:55:45.117234Z","shell.execute_reply.started":"2024-12-07T10:55:05.341067Z","shell.execute_reply":"2024-12-07T10:55:45.116155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def pair_corr(input_data: tuple):\n    \"\"\"\n    Calculates the correlation of two features \n    and writes the result to the correlation matrix\n    \n    Parameters\n    ----------\n    input_data: a tuple of three attributes: \n        df      : a reference to a dataframe\n        comb    : indices of a pair of attributes (tuple)\n        corr_arr: a reference to a correlation matrix\n        prefix  : feature name prefix ('feature_' | 'responder_')\n        width   : width allocated for index (2 for '01'-'09' else 1)\n    \n    \"\"\"\n    \n    df, comb, corr_arr, prefix, width = input_data\n    i, j = comb\n    coef = df.select(pl.corr(*[f'{prefix}{numb:0{width}d}' for numb in (i, j)])).item()\n    corr_arr[i,j] = corr_arr[j,i] = coef","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:55:45.118415Z","iopub.execute_input":"2024-12-07T10:55:45.118764Z","iopub.status.idle":"2024-12-07T10:55:45.125437Z","shell.execute_reply.started":"2024-12-07T10:55:45.118700Z","shell.execute_reply":"2024-12-07T10:55:45.124094Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Correlation of responders","metadata":{}},{"cell_type":"markdown","source":"For the sake of clarity, let's start with **responders**, since there are fewer of them","metadata":{}},{"cell_type":"code","source":"# preparing a template for the correlation matrix and a list of combinations of responders numbers\nresp_corr = np.zeros([RESPONDERS_NUMBER]*2)\nnp.fill_diagonal(resp_corr, 1)\ncombs = list(combinations(range(RESPONDERS_NUMBER), 2))\n\npprint(['responders number combinations:', combs, f'number of combinations: {len(combs)}'], compact=True, width=100)\nresp_corr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:55:45.126936Z","iopub.execute_input":"2024-12-07T10:55:45.127299Z","iopub.status.idle":"2024-12-07T10:55:45.156452Z","shell.execute_reply.started":"2024-12-07T10:55:45.127264Z","shell.execute_reply":"2024-12-07T10:55:45.155087Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"_ = thread_map(pair_corr, \n               [(df, comb, resp_corr, 'responder_', 1) for comb in combs], \n               tqdm_class=async_tqdm, total=len(combs), ncols=100\n              )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:55:45.157993Z","iopub.execute_input":"2024-12-07T10:55:45.158421Z","iopub.status.idle":"2024-12-07T10:55:47.017475Z","shell.execute_reply.started":"2024-12-07T10:55:45.158372Z","shell.execute_reply":"2024-12-07T10:55:47.016194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with np.printoptions(precision=5, suppress=True, linewidth=130):\n    display(resp_corr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:55:47.019134Z","iopub.execute_input":"2024-12-07T10:55:47.019581Z","iopub.status.idle":"2024-12-07T10:55:47.029963Z","shell.execute_reply.started":"2024-12-07T10:55:47.019532Z","shell.execute_reply":"2024-12-07T10:55:47.028681Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Correlations of features","metadata":{}},{"cell_type":"markdown","source":"Now let's do the same with **features**","metadata":{}},{"cell_type":"code","source":"feat_corr = np.zeros([FEATURES_NUMBER]*2)\nnp.fill_diagonal(feat_corr, 1)\ncombs = list(combinations(range(FEATURES_NUMBER), 2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:55:47.031127Z","iopub.execute_input":"2024-12-07T10:55:47.031420Z","iopub.status.idle":"2024-12-07T10:55:47.044810Z","shell.execute_reply.started":"2024-12-07T10:55:47.031393Z","shell.execute_reply":"2024-12-07T10:55:47.043798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"_ = thread_map(pair_corr, \n               [(df, comb, feat_corr, 'feature_', 2) for comb in combs], \n               tqdm_class=async_tqdm, total=len(combs), ncols=100\n              )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:55:47.046279Z","iopub.execute_input":"2024-12-07T10:55:47.046621Z","iopub.status.idle":"2024-12-07T11:02:46.961374Z","shell.execute_reply.started":"2024-12-07T10:55:47.046590Z","shell.execute_reply":"2024-12-07T11:02:46.959271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(feat_corr.shape)\nwith np.printoptions(precision=6, suppress=True, linewidth=130):\n    display(feat_corr[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:02:46.963954Z","iopub.execute_input":"2024-12-07T11:02:46.964468Z","iopub.status.idle":"2024-12-07T11:02:46.980051Z","shell.execute_reply.started":"2024-12-07T11:02:46.964431Z","shell.execute_reply":"2024-12-07T11:02:46.978788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# save the data to disk\nnp.save('resp_corr.npy',resp_corr)\nnp.save('feat_corr.npy', feat_corr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:02:46.981842Z","iopub.execute_input":"2024-12-07T11:02:46.982187Z","iopub.status.idle":"2024-12-07T11:02:46.994980Z","shell.execute_reply.started":"2024-12-07T11:02:46.982154Z","shell.execute_reply":"2024-12-07T11:02:46.993767Z"}},"outputs":[],"execution_count":null}]}