{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\n\nfrom collections import defaultdict\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom scipy.stats import spearmanr\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.linear_model import Ridge\nfrom sklearn.metrics import r2_score","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:23:26.013836Z","iopub.execute_input":"2024-11-13T23:23:26.015073Z","iopub.status.idle":"2024-11-13T23:23:28.054829Z","shell.execute_reply.started":"2024-11-13T23:23:26.015012Z","shell.execute_reply":"2024-11-13T23:23:28.053515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = []\nfor partition_id in range(10):\n    path = \"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={}/part-0.parquet\".format(partition_id)\n    df = pd.read_parquet(path, \n                         columns = [\n                             'symbol_id', 'date_id','time_id', 'responder_6', 'weight',\n                             'feature_05', 'feature_06'\n                         ])\n    \n    #df = df[['symbol_id', 'date_id','time_id', 'responder_6']]\n    df.rename(columns={\n        'responder_6': 'target'\n    }, inplace=True)\n    train_df.append(df)\n\ntrain_df = pd.concat(train_df)\ntrain_df = train_df.sort_values(['symbol_id', 'date_id', 'time_id'])\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:23:28.057588Z","iopub.execute_input":"2024-11-13T23:23:28.058440Z","iopub.status.idle":"2024-11-13T23:23:50.852758Z","shell.execute_reply.started":"2024-11-13T23:23:28.058373Z","shell.execute_reply":"2024-11-13T23:23:50.851186Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def evaluation_metric(ytrue,ypred,w):\n    num = (w*(ytrue-ypred)**2).sum()\n    den = (w*(ytrue**2)).sum()\n    return (1-num/den)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:23:50.854548Z","iopub.execute_input":"2024-11-13T23:23:50.855068Z","iopub.status.idle":"2024-11-13T23:23:50.863021Z","shell.execute_reply.started":"2024-11-13T23:23:50.855017Z","shell.execute_reply":"2024-11-13T23:23:50.861497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SYMBOL_ID=10\nsymbol_df = train_df[train_df.symbol_id == SYMBOL_ID]\ndateid_list = sorted(symbol_df.date_id.unique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:23:50.866557Z","iopub.execute_input":"2024-11-13T23:23:50.867121Z","iopub.status.idle":"2024-11-13T23:23:51.010932Z","shell.execute_reply.started":"2024-11-13T23:23:50.867067Z","shell.execute_reply":"2024-11-13T23:23:51.009423Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Heuristic-1\n\nusing the previous days last target value as the response to the present day.","metadata":{}},{"cell_type":"code","source":"target_last_df = symbol_df.groupby(['symbol_id', 'date_id'])[['target']].last().reset_index().rename(columns={\n    'target': 'last_target'\n})\ntarget_last_df = target_last_df.sort_values(['symbol_id', 'date_id'])\ntarget_last_df['last_target_lag1'] = target_last_df.groupby('symbol_id')[['last_target']].shift(1)\n\ntrain_df = symbol_df.merge(target_last_df)\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:23:51.013406Z","iopub.execute_input":"2024-11-13T23:23:51.014062Z","iopub.status.idle":"2024-11-13T23:23:51.303454Z","shell.execute_reply.started":"2024-11-13T23:23:51.013998Z","shell.execute_reply":"2024-11-13T23:23:51.302142Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ytrue = train_df[train_df.date_id>0]['target'].values\nypred = train_df[train_df.date_id>0]['last_target_lag1'].values\nweights = train_df[train_df.date_id>0]['weight'].values\n\nevaluation_metric(ytrue, ypred, weights)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:23:51.304980Z","iopub.execute_input":"2024-11-13T23:23:51.305412Z","iopub.status.idle":"2024-11-13T23:23:51.511607Z","shell.execute_reply.started":"2024-11-13T23:23:51.305369Z","shell.execute_reply":"2024-11-13T23:23:51.509641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ytrue = train_df[(train_df.date_id>0) & \n                (train_df.time_id<200)]['target'].values\n\nypred = train_df[(train_df.date_id>0) & \n                (train_df.time_id<200)]['last_target_lag1'].values\n\n\n\nweights = train_df[(train_df.date_id>0) & \n                (train_df.time_id<200)]['weight'].values\n\nevaluation_metric(ytrue, ypred, weights)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:23:51.513826Z","iopub.execute_input":"2024-11-13T23:23:51.514366Z","iopub.status.idle":"2024-11-13T23:23:51.589596Z","shell.execute_reply.started":"2024-11-13T23:23:51.514282Z","shell.execute_reply":"2024-11-13T23:23:51.587914Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ytrue = train_df[(train_df.date_id>0) & \n                (train_df.time_id<100)]['target'].values\n\nypred = train_df[(train_df.date_id>0) & \n                (train_df.time_id<100)]['last_target_lag1'].values\n\nweights = train_df[(train_df.date_id>0) & \n                (train_df.time_id<100)]['weight'].values\n\nevaluation_metric(ytrue, ypred, weights)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:23:51.591531Z","iopub.execute_input":"2024-11-13T23:23:51.592102Z","iopub.status.idle":"2024-11-13T23:23:51.648277Z","shell.execute_reply.started":"2024-11-13T23:23:51.592039Z","shell.execute_reply":"2024-11-13T23:23:51.647012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ytrue = train_df[(train_df.date_id>0) & \n                (train_df.time_id>600)]['target'].values\n\nypred = train_df[(train_df.date_id>0) & \n                (train_df.time_id>600)]['last_target_lag1'].values\n\nweights = train_df[(train_df.date_id>0) & \n                (train_df.time_id>600)]['weight'].values\n\nevaluation_metric(ytrue, ypred, weights)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:23:51.650028Z","iopub.execute_input":"2024-11-13T23:23:51.650480Z","iopub.status.idle":"2024-11-13T23:23:51.739845Z","shell.execute_reply.started":"2024-11-13T23:23:51.650434Z","shell.execute_reply":"2024-11-13T23:23:51.738642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(train_df[(train_df.date_id==0)])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:23:51.744783Z","iopub.execute_input":"2024-11-13T23:23:51.745288Z","iopub.status.idle":"2024-11-13T23:23:51.756550Z","shell.execute_reply.started":"2024-11-13T23:23:51.745240Z","shell.execute_reply":"2024-11-13T23:23:51.755155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ytrue = train_df[(train_df.date_id>0) & \n                (train_df.time_id>750)]['target'].values\n\nypred = train_df[(train_df.date_id>0) & \n                (train_df.time_id>750)]['last_target_lag1'].values\n\nweights = train_df[(train_df.date_id>0) & \n                (train_df.time_id>750)]['weight'].values\n\n\nevaluation_metric(ytrue, ypred, weights)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:23:51.758464Z","iopub.execute_input":"2024-11-13T23:23:51.759050Z","iopub.status.idle":"2024-11-13T23:23:51.824594Z","shell.execute_reply.started":"2024-11-13T23:23:51.758987Z","shell.execute_reply":"2024-11-13T23:23:51.823179Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATEID=8\nfig, ax = plt.subplots(1, 2, figsize=(14, 4))\nsns.lineplot(data=target_last_df, x='date_id', y='last_target', ax=ax[0])\nsns.lineplot(\n    data=train_df[train_df.date_id==DATEID], \n    x='time_id',\n    y='target', \n    ax=ax[1])\nsns.lineplot(\n    data=train_df[train_df.date_id==DATEID], \n    x='time_id',\n    y='last_target_lag1',\n    ax=ax[1])\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:23:51.826347Z","iopub.execute_input":"2024-11-13T23:23:51.826808Z","iopub.status.idle":"2024-11-13T23:23:52.472601Z","shell.execute_reply.started":"2024-11-13T23:23:51.826753Z","shell.execute_reply":"2024-11-13T23:23:52.471051Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Observations**\n1. A simple last target prediction has given an weighted R2 evaluation: -0.200\n2. The same heuristic with first 200 samples : -0.1326\n3. The same heuristic with first 100 samples : -0.102\n4. with the samples with time_id more than 600 : -0.199\n5. with the samples with time_id more than 800 : -0.173\n\nthis suggest last sample of previous day has some influence on the next day initial samples than latter.\n\n**Future Work**:\n\n(we need to check the analysis not for the single last sample, but some set of samples with the next day first few initial sample)","metadata":{}},{"cell_type":"markdown","source":"#### Analysis on the target distribution and its change over the time.","metadata":{}},{"cell_type":"code","source":"dateid_list=np.arange(148, 158)\n\nfig, ax = plt.subplots(1, 2, figsize=(14, 5))\nfor date_id in dateid_list:\n    df = symbol_df[symbol_df.date_id == date_id]\n    sns.histplot(data=symbol_df[symbol_df.date_id == date_id], x='target', label=date_id, ax=ax[0])\nsns.boxplot(data=symbol_df[symbol_df.date_id.isin(dateid_list)], x='date_id', y='target', ax=ax[1])\n\nax[0].set_title(\"target historgram\")\nax[1].set_title(\"target boxplot\")\n\nax[0].legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:23:52.474374Z","iopub.execute_input":"2024-11-13T23:23:52.474841Z","iopub.status.idle":"2024-11-13T23:23:54.585492Z","shell.execute_reply.started":"2024-11-13T23:23:52.474795Z","shell.execute_reply":"2024-11-13T23:23:54.583998Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dateid_list=np.arange(148, 158)\n\nfig, ax = plt.subplots(1, 2, figsize=(14, 5))\nfor date_id in dateid_list:\n    df = symbol_df[symbol_df.date_id == date_id]\n    sns.histplot(data=symbol_df[symbol_df.date_id == date_id], x='feature_05', label=date_id, ax=ax[0])\nsns.boxplot(data=symbol_df[symbol_df.date_id.isin(dateid_list)], x='date_id', y='feature_05', ax=ax[1])\n\nax[0].set_title(\"feature_05 historgram\")\nax[1].set_title(\"feature_05 boxplot\")\n\nax[0].legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:23:54.587327Z","iopub.execute_input":"2024-11-13T23:23:54.587911Z","iopub.status.idle":"2024-11-13T23:23:56.368077Z","shell.execute_reply.started":"2024-11-13T23:23:54.587847Z","shell.execute_reply":"2024-11-13T23:23:56.366392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dateid_list=np.arange(148, 158)\n\nfig, ax = plt.subplots(1, 2, figsize=(14, 5))\nfor date_id in dateid_list:\n    df = symbol_df[symbol_df.date_id == date_id]\n    sns.histplot(data=symbol_df[symbol_df.date_id == date_id], x='feature_06', label=date_id, ax=ax[0])\nsns.boxplot(data=symbol_df[symbol_df.date_id.isin(dateid_list)], x='date_id', y='feature_06', ax=ax[1])\n\nax[0].set_title(\"feature_06 historgram\")\nax[1].set_title(\"feature_06 boxplot\")\n\nax[0].legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:23:56.369915Z","iopub.execute_input":"2024-11-13T23:23:56.370408Z","iopub.status.idle":"2024-11-13T23:23:59.334219Z","shell.execute_reply.started":"2024-11-13T23:23:56.370358Z","shell.execute_reply":"2024-11-13T23:23:59.332690Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(13, 5))\nfor date_id in dateid_list[:6]:\n    sns.scatterplot(data=symbol_df[symbol_df.date_id==date_id], x='feature_05', y='target', label=date_id)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:23:59.336274Z","iopub.execute_input":"2024-11-13T23:23:59.336863Z","iopub.status.idle":"2024-11-13T23:24:00.036884Z","shell.execute_reply.started":"2024-11-13T23:23:59.336800Z","shell.execute_reply":"2024-11-13T23:24:00.035219Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Observations**\n\n","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(2,2, figsize=(12,6), sharex=True, sharey=True)\n\nfig.suptitle(\"feature_05 (vs) target\")\nsns.lineplot(data=symbol_df[symbol_df.date_id==148], x='feature_05', y='target', ax=ax[0][0])\nsns.lineplot(data=symbol_df[symbol_df.date_id==149], x='feature_05', y='target', ax=ax[0][1])\n\nsns.lineplot(data=symbol_df[symbol_df.date_id==150], x='feature_05', y='target', ax=ax[1][0])\nsns.lineplot(data=symbol_df[symbol_df.date_id==151], x='feature_05', y='target', ax=ax[1][1])\n\nfor i in [0,1]:\n    for j in [0, 1]:\n        ax[i][j].vlines(ymin=-2, ymax=2, x=0, color='r', linestyle='--')\n        ax[i][j].hlines(xmin=-3, xmax=4, y=0, color='r', linestyle='--')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:00.038798Z","iopub.execute_input":"2024-11-13T23:24:00.039393Z","iopub.status.idle":"2024-11-13T23:24:01.071769Z","shell.execute_reply.started":"2024-11-13T23:24:00.039327Z","shell.execute_reply":"2024-11-13T23:24:01.070147Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(2,2, figsize=(12,6), sharex=True, sharey=True)\n\nfig.suptitle(\"feature_06 (vs) target\")\nsns.lineplot(data=symbol_df[symbol_df.date_id==148], x='feature_06', y='target', ax=ax[0][0])\nsns.lineplot(data=symbol_df[symbol_df.date_id==149], x='feature_06', y='target', ax=ax[0][1])\n\nsns.lineplot(data=symbol_df[symbol_df.date_id==150], x='feature_06', y='target', ax=ax[1][0])\nsns.lineplot(data=symbol_df[symbol_df.date_id==151], x='feature_06', y='target', ax=ax[1][1])\n\nfor i in [0,1]:\n    for j in [0, 1]:\n        ax[i][j].vlines(ymin=-2, ymax=2, x=0, color='r', linestyle='--')\n        ax[i][j].hlines(xmin=-3, xmax=4, y=0, color='r', linestyle='--')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:01.074912Z","iopub.execute_input":"2024-11-13T23:24:01.076085Z","iopub.status.idle":"2024-11-13T23:24:02.176829Z","shell.execute_reply.started":"2024-11-13T23:24:01.076012Z","shell.execute_reply":"2024-11-13T23:24:02.175109Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Observation**\n1. target distribution changes from one day to the other.\n2. feature distribution chages from one day to the other.\n3. from the sample graphs above, looks there is some pattern in the extremes, need to analyze furthur to understand if the sample pattern is valid for larger subsets.","metadata":{}},{"cell_type":"markdown","source":"**Heuristic Baseline for target variable**\n\nSimple statistics at quantiles (10%, 30%, 50%, 70%, 90%) are used as target for the next day.","metadata":{}},{"cell_type":"code","source":"stat_df = symbol_df[['symbol_id', 'date_id']].drop_duplicates()\nfor q in [0.1, 0.3, 0.5, 0.7, 0.9]:\n    qdf = symbol_df.groupby(['symbol_id', 'date_id'])[['target']].quantile(q).rename(columns={\n        'target':'target_{}'.format(q)\n    }).reset_index()\n    stat_df = stat_df.merge(qdf)\ntrain_df2 = symbol_df[['symbol_id', 'date_id','time_id', 'target']].merge(stat_df)\n\ntrain_df2.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:02.178990Z","iopub.execute_input":"2024-11-13T23:24:02.179601Z","iopub.status.idle":"2024-11-13T23:24:03.564834Z","shell.execute_reply.started":"2024-11-13T23:24:02.179523Z","shell.execute_reply":"2024-11-13T23:24:03.563443Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_heuristic_metrics():\n    heurist_df = []\n    \n    for heuristic_column in ['target_0.1', 'target_0.3', 'target_0.5', 'target_0.7', 'target_0.9']:\n        for LAG in [1]:\n            PREDICTED_COLUMN=heuristic_column\n            stat_df['lag_'+heuristic_column] = stat_df.groupby('symbol_id')[[heuristic_column]].shift(LAG)\n            df = symbol_df[['symbol_id', 'date_id','time_id', 'weight', 'target']].merge(stat_df[['symbol_id', 'date_id']+ ['lag_'+heuristic_column, heuristic_column]])\n            df = df[df.date_id > 0]\n            \n            heurist_df.append({\n                'heuristic_column': heuristic_column,\n                \"lag\": LAG,\n                'metric_lag': evaluation_metric(df.target.values, df['lag_'+heuristic_column], df.weight.values),\n                'corr_lag': np.corrcoef(df.target.values, df['lag_'+heuristic_column])[0, 1],\n                'metric': evaluation_metric(df.target.values, df[heuristic_column], df.weight.values),\n                'corr': np.corrcoef(df.target.values, df[heuristic_column])[0, 1],\n            })\n    heuristic_df = pd.DataFrame.from_dict(heurist_df)\n    return heuristic_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:03.566594Z","iopub.execute_input":"2024-11-13T23:24:03.567053Z","iopub.status.idle":"2024-11-13T23:24:03.580749Z","shell.execute_reply.started":"2024-11-13T23:24:03.567005Z","shell.execute_reply":"2024-11-13T23:24:03.578778Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"heuristic_df = get_heuristic_metrics()\nheuristic_df = heuristic_df.sort_values('metric', ascending=False)\nheuristic_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:03.583125Z","iopub.execute_input":"2024-11-13T23:24:03.583935Z","iopub.status.idle":"2024-11-13T23:24:05.359883Z","shell.execute_reply.started":"2024-11-13T23:24:03.583868Z","shell.execute_reply":"2024-11-13T23:24:05.357951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(14, 4))\nsns.barplot(data=heuristic_df, x='heuristic_column', y='metric', ax=ax[0])\nsns.barplot(data=heuristic_df, x='heuristic_column', y='corr', ax=ax[1])\n\nax[0].set_title(\"evaluation metric\")\nax[1].set_title(\"correlation\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:05.362124Z","iopub.execute_input":"2024-11-13T23:24:05.362738Z","iopub.status.idle":"2024-11-13T23:24:06.320570Z","shell.execute_reply.started":"2024-11-13T23:24:05.362684Z","shell.execute_reply":"2024-11-13T23:24:06.319093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(14, 4))\nsns.heatmap(train_df2[['target'] + ['target_0.1', 'target_0.3', \n                                    'target_0.5', 'target_0.7', 'target_0.9']].corr(), annot=True, fmt='.2f', ax=ax[0])\nsns.scatterplot(data=train_df2, x='target_0.5', y='target', ax=ax[1])\nax[0].set_title(\"same day quantiles (vs) target\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:06.322170Z","iopub.execute_input":"2024-11-13T23:24:06.322633Z","iopub.status.idle":"2024-11-13T23:24:10.820798Z","shell.execute_reply.started":"2024-11-13T23:24:06.322588Z","shell.execute_reply":"2024-11-13T23:24:10.818943Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**observations**\n1. Median value of the previous day explained better than any other quantiles and last value.\n2. It is as expected that, as we move to the extreme value as the representative value, error increases.","metadata":{}},{"cell_type":"markdown","source":"### Analysis on the target quantiles","metadata":{}},{"cell_type":"code","source":"print(\"number of records:\", len(stat_df))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:10.822528Z","iopub.execute_input":"2024-11-13T23:24:10.823114Z","iopub.status.idle":"2024-11-13T23:24:10.831725Z","shell.execute_reply.started":"2024-11-13T23:24:10.823046Z","shell.execute_reply":"2024-11-13T23:24:10.830090Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(14,4))\nfor column in ['target_0.1', 'target_0.3', 'target_0.5', 'target_0.7', 'target_0.9']:\n    sns.lineplot(data=stat_df.head(1200), x='date_id', y=column, label=column)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:10.833660Z","iopub.execute_input":"2024-11-13T23:24:10.834244Z","iopub.status.idle":"2024-11-13T23:24:11.462005Z","shell.execute_reply.started":"2024-11-13T23:24:10.834179Z","shell.execute_reply":"2024-11-13T23:24:11.460571Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ax = sns.heatmap(stat_df[['target_0.1', 'target_0.3', 'target_0.5', 'target_0.7', 'target_0.9']].corr(), annot=True, fmt='.2f')\nax.xaxis.tick_top()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:11.464094Z","iopub.execute_input":"2024-11-13T23:24:11.464719Z","iopub.status.idle":"2024-11-13T23:24:12.100205Z","shell.execute_reply.started":"2024-11-13T23:24:11.464655Z","shell.execute_reply":"2024-11-13T23:24:12.098692Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(13, 4))\nsns.scatterplot(\n    data=stat_df[(stat_df['target_0.1'] > -2)],\n    x='target_0.1',\n    y='target_0.3',\n    ax=ax[0]\n)\n\nsns.scatterplot(\n    data=stat_df[(stat_df['target_0.1'] > -2)],\n    x='target_0.1',\n    y='target_0.5',\n    ax=ax[1]\n)\nax[0].set_title(\"target_0.3\")\nax[1].set_title(\"target_0.5\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:12.102149Z","iopub.execute_input":"2024-11-13T23:24:12.102740Z","iopub.status.idle":"2024-11-13T23:24:12.661988Z","shell.execute_reply.started":"2024-11-13T23:24:12.102673Z","shell.execute_reply":"2024-11-13T23:24:12.660234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(13, 4))\nsns.scatterplot(\n    data=stat_df[(stat_df['target_0.1'] > -2)],\n    x='target_0.1',\n    y='target_0.7',\n    ax=ax[0]\n)\n\nsns.scatterplot(\n    data=stat_df[(stat_df['target_0.1'] > -2)],\n    x='target_0.1',\n    y='target_0.9',\n    ax=ax[1]\n)\nax[0].set_title(\"target_0.7\")\nax[1].set_title(\"target_0.9\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:12.673200Z","iopub.execute_input":"2024-11-13T23:24:12.673763Z","iopub.status.idle":"2024-11-13T23:24:13.328088Z","shell.execute_reply.started":"2024-11-13T23:24:12.673716Z","shell.execute_reply":"2024-11-13T23:24:13.326217Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**observation**\n\n1. It is interesting to see that, when comparing other percentile with 10th percentile, have positive correlation with 30th, 50th percentile.\n2. But correlation is negative with 90th percentile, i.e. as the 10th percentile increases 90th percentile decreases.\n3. **Hypothesis**: variance of days distribution will reduce as 10th percentile increases.\n\n**Note:** \nMake furthur analysis in this direction if this can be generalized.","metadata":{}},{"cell_type":"markdown","source":"#### lets check if the hypothesis of variance redcution with 10th percentile increase .","metadata":{}},{"cell_type":"code","source":"df = symbol_df.groupby(['symbol_id', 'date_id'])[['target']].std().rename(columns={'target': 'target_std'}).reset_index()\ndf = df.merge(stat_df)\ndf[['symbol_id', 'date_id', \n    'target_std', 'target_0.1', 'target_0.3', 'target_0.5', 'target_0.7', 'target_0.9']].head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:13.330593Z","iopub.execute_input":"2024-11-13T23:24:13.331078Z","iopub.status.idle":"2024-11-13T23:24:13.452502Z","shell.execute_reply.started":"2024-11-13T23:24:13.331030Z","shell.execute_reply":"2024-11-13T23:24:13.450788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nsns.histplot(\n    data=symbol_df[symbol_df.date_id==8],\n    x='target',\n    label=\"date=8\"\n)\n\nsns.histplot(\n    data=symbol_df[symbol_df.date_id==0],\n    x='target',\n    label=\"date=0\"\n)\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:13.454589Z","iopub.execute_input":"2024-11-13T23:24:13.455197Z","iopub.status.idle":"2024-11-13T23:24:13.989056Z","shell.execute_reply.started":"2024-11-13T23:24:13.455140Z","shell.execute_reply":"2024-11-13T23:24:13.987602Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 4))\nax = sns.heatmap(\n    df[['target_std', 'target_0.1', 'target_0.3', 'target_0.5', 'target_0.7', 'target_0.9']].corr(),\n    annot=True,\n    fmt='.2f'\n)\n\nax.xaxis.tick_top()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:13.991352Z","iopub.execute_input":"2024-11-13T23:24:13.991805Z","iopub.status.idle":"2024-11-13T23:24:14.487859Z","shell.execute_reply.started":"2024-11-13T23:24:13.991759Z","shell.execute_reply":"2024-11-13T23:24:14.486513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(15, 4))\nsns.lineplot(\n    data=df,\n    x='date_id',\n    y='target_std'\n)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:14.490035Z","iopub.execute_input":"2024-11-13T23:24:14.490632Z","iopub.status.idle":"2024-11-13T23:24:14.854746Z","shell.execute_reply.started":"2024-11-13T23:24:14.490568Z","shell.execute_reply":"2024-11-13T23:24:14.853400Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xstd = df.target_std.values\n\nprint(np.corrcoef(xstd[:-1], xstd[1:])[0][1])\nsns.scatterplot(\n    x=xstd[:-1],\n    y=xstd[1:]\n)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:14.856552Z","iopub.execute_input":"2024-11-13T23:24:14.857027Z","iopub.status.idle":"2024-11-13T23:24:15.138914Z","shell.execute_reply.started":"2024-11-13T23:24:14.856970Z","shell.execute_reply":"2024-11-13T23:24:15.137383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"date_ids = symbol_df.date_id.unique()[100:]\nprint(date_ids[:10])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:15.140838Z","iopub.execute_input":"2024-11-13T23:24:15.141346Z","iopub.status.idle":"2024-11-13T23:24:15.158631Z","shell.execute_reply.started":"2024-11-13T23:24:15.141277Z","shell.execute_reply":"2024-11-13T23:24:15.157001Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### ACF for quantile statistics","metadata":{}},{"cell_type":"code","source":"for target_col in ['target_0.1', 'target_0.3', 'target_0.5', 'target_0.7', 'target_0.9']:\n    x = stat_df[stat_df[target_col]>-3][target_col].values\n    \n    for LAG in [1 ]:\n        xlag = x[:-LAG]\n        xcur = x[LAG:]\n        fig, ax = plt.subplots(1, 2, figsize=(12, 3))\n        sns.scatterplot(x=xlag, y=xcur, ax=ax[0])\n        sns.heatmap(np.corrcoef(xlag, xcur), ax=ax[1], annot=True, fmt='.2f', \n                    yticklabels=['xlag', 'xcur'], \n                    xticklabels=['xlag', 'xcur'])\n        \n        fig.suptitle(target_col+ \" | LAG:{}\".format(LAG))\n        plt.show()\nstat_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:15.160516Z","iopub.execute_input":"2024-11-13T23:24:15.160989Z","iopub.status.idle":"2024-11-13T23:24:18.105715Z","shell.execute_reply.started":"2024-11-13T23:24:15.160931Z","shell.execute_reply":"2024-11-13T23:24:18.104373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from statsmodels.graphics.tsaplots import plot_acf\n\nfor target_col in ['target_0.1', 'target_0.3', 'target_0.5', 'target_0.7', 'target_0.9']:\n    fig,ax = plt.subplots(1,2,figsize=(12, 4))\n    plot_acf(stat_df[target_col], ax=ax[0])\n    sns.lineplot(data=stat_df.head(1200), x='date_id', y=target_col)\n    fig.suptitle(target_col)\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:18.107614Z","iopub.execute_input":"2024-11-13T23:24:18.108049Z","iopub.status.idle":"2024-11-13T23:24:21.682041Z","shell.execute_reply.started":"2024-11-13T23:24:18.108003Z","shell.execute_reply":"2024-11-13T23:24:21.680642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(14,5))\nsns.lineplot(data=stat_df.head(200), x='date_id', y='target_0.5')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:21.683808Z","iopub.execute_input":"2024-11-13T23:24:21.684529Z","iopub.status.idle":"2024-11-13T23:24:22.025515Z","shell.execute_reply.started":"2024-11-13T23:24:21.684466Z","shell.execute_reply":"2024-11-13T23:24:22.024017Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dateid_list=np.arange(148, 158)\nfor date_id in dateid_list:\n    df = symbol_df[symbol_df.date_id == date_id]\n    #sns.histplot(data=symbol_df[symbol_df.date_id == date_id], x='target', label=date_id)\nsns.boxplot(data=symbol_df[symbol_df.date_id.isin(dateid_list)], x='date_id', y='target')    \nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:22.027384Z","iopub.execute_input":"2024-11-13T23:24:22.027851Z","iopub.status.idle":"2024-11-13T23:24:22.512854Z","shell.execute_reply.started":"2024-11-13T23:24:22.027804Z","shell.execute_reply":"2024-11-13T23:24:22.511378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.lineplot(data=symbol_df[symbol_df.date_id == 148], x='time_id', y='target')\nsns.lineplot(data=symbol_df[symbol_df.date_id == 149], x='time_id', y='target')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:22.514954Z","iopub.execute_input":"2024-11-13T23:24:22.515477Z","iopub.status.idle":"2024-11-13T23:24:22.886502Z","shell.execute_reply.started":"2024-11-13T23:24:22.515426Z","shell.execute_reply":"2024-11-13T23:24:22.884369Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### lets see the replationship between feature variables and the target variable","metadata":{}},{"cell_type":"markdown","source":"1. Correlation between the feature and target vary between consecutive days.\n2. If there is no influence of the other variables, using model developed for previous day may not be su","metadata":{}},{"cell_type":"code","source":"for LAG in [0, 1]:\n    for FEATURE_COLUMN in ['feature_05', 'feature_06']:\n        corr_list = []\n        for date_id in sorted(symbol_df.date_id.unique()):\n            symbol_subset_df = symbol_df[symbol_df.date_id==date_id]\n            x = symbol_subset_df[FEATURE_COLUMN].values\n            y = symbol_subset_df['target'].values\n            if LAG !=0:\n                x = x[:-LAG]; y=y[LAG:]\n            corr_list.append({\n                'date_id': date_id,\n                'corr': np.corrcoef(x, y)[0, 1]\n            })\n            \n        corr_list = pd.DataFrame.from_dict(corr_list)\n        print(corr_list.head(4))\n        fig, ax = plt.subplots(1, 2, figsize=(15, 5))\n        fig.suptitle(FEATURE_COLUMN+\" (vs) target correlation - {}\".format(LAG))\n        sns.lineplot(data=corr_list, x='date_id', y='corr', ax=ax[0])\n        sns.histplot(corr_list['corr'].values, ax=ax[1])\n        plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:22.888940Z","iopub.execute_input":"2024-11-13T23:24:22.889581Z","iopub.status.idle":"2024-11-13T23:24:36.540336Z","shell.execute_reply.started":"2024-11-13T23:24:22.889517Z","shell.execute_reply":"2024-11-13T23:24:36.538927Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(14, 4))\nfor date_id in [2, 3]:\n    df = symbol_df[symbol_df.date_id==date_id]\n    sns.scatterplot(\n        data=df,\n        x='feature_05',\n        y='target',\n        label='date={}'.format(date_id),\n        ax=ax[0]\n    )\n    sns.histplot(\n        data=df,\n        x='target', \n        label='date={}'.format(date_id),\n        ax=ax[1]\n    )\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:36.542251Z","iopub.execute_input":"2024-11-13T23:24:36.542843Z","iopub.status.idle":"2024-11-13T23:24:37.596224Z","shell.execute_reply.started":"2024-11-13T23:24:36.542780Z","shell.execute_reply":"2024-11-13T23:24:37.594825Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(14, 5))\nfor date_id in [2, 3]:\n    df = symbol_df[symbol_df.date_id==date_id]\n    df = df.sort_values('feature_05')\n    \n    sns.scatterplot(\n        data=df,\n        x='feature_05',\n        y='target',\n        label='date={}'.format(date_id)\n    )\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:37.597926Z","iopub.execute_input":"2024-11-13T23:24:37.598391Z","iopub.status.idle":"2024-11-13T23:24:37.989653Z","shell.execute_reply.started":"2024-11-13T23:24:37.598335Z","shell.execute_reply":"2024-11-13T23:24:37.988360Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"symbol_df[symbol_df.date_id.isin([2,3])].groupby('date_id')[['target']].std()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:37.991471Z","iopub.execute_input":"2024-11-13T23:24:37.991987Z","iopub.status.idle":"2024-11-13T23:24:38.014678Z","shell.execute_reply.started":"2024-11-13T23:24:37.991936Z","shell.execute_reply":"2024-11-13T23:24:38.013427Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stat_df[stat_df.date_id.isin([2,3])]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:38.016504Z","iopub.execute_input":"2024-11-13T23:24:38.016946Z","iopub.status.idle":"2024-11-13T23:24:38.042916Z","shell.execute_reply.started":"2024-11-13T23:24:38.016899Z","shell.execute_reply":"2024-11-13T23:24:38.041407Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**observation**\n1. correlation between the feature and target is not linear and ossicllating between positive and negative.\n2. the negative correlation for date=3 might likely be due to extreme values of feature_05","metadata":{}},{"cell_type":"markdown","source":"### lets add new heuristic based on feature_05\n1. split feature_05 to quantiles, and the targets based on previous day.\n2. assign the averaged target values in the range to the next day. ","metadata":{}},{"cell_type":"code","source":"date_list = sorted(symbol_df.date_id.unique())\n\ndef get_previous_metrics(df):\n    df = df.sort_values('feature_05')\n    x = df['feature_05'].values\n    y = df['target'].values\n\n    xq10 = np.quantile(x, 0.1)\n    xq30 = np.quantile(x, 0.3)\n    xq50 = np.quantile(x, 0.5)\n    xq70 = np.quantile(x, 0.7)\n    xq90 = np.quantile(x, 0.9)\n\n    ymean_10 = y[x<xq10].mean()\n    ymean_30 = y[(x>xq10) & (x<=xq30)].mean()\n    ymean_50 = y[(x>xq30) & (x<=xq50)].mean()\n    ymean_70 = y[(x>xq50) & (x<=xq70)].mean()\n    ymean_90 = y[(x>xq90)].mean()\n\n    ymedian_10 = np.median(y[x<=xq10])\n    ymedian_30 = np.median(y[(x>xq10) & (x<=xq30)])\n    ymedian_50 = np.median(y[(x>xq30) & (x<=xq50)])\n    ymedian_70 = np.median(y[(x>xq50) & (x<=xq70)])\n    ymedian_90 = np.median(y[(x>xq90)]) \n    \n    quantile_map = {\n        'q10': (xq10, ymean_10, ymedian_10),\n        'q30': (xq30, ymean_30, ymedian_30),\n        'q50': (xq50, ymean_50, ymedian_50),\n        'q70': (xq70, ymean_70, ymedian_70),\n        'q90': (xq90, ymean_90, ymedian_90)\n    }\n    return quantile_map\n\n\ndef get_mean_estimate(row, quantile_map):\n    x = row['feature_05']\n\n    q10 = quantile_map['q10'][0]\n    q30 = quantile_map['q30'][0]\n    q50 = quantile_map['q50'][0]\n    q70 = quantile_map['q70'][0]\n    q90 = quantile_map['q90'][0]\n    \n    \n    if x<=q10:\n        return quantile_map['q10'][1]\n    elif q10<=x and x<=q30:\n        return quantile_map['q30'][1]\n    elif q30<=x and x<=q50:\n        return quantile_map['q50'][1]\n    elif q50<=x and x<=q70:\n        return quantile_map['q70'][1]\n    else:\n        return quantile_map['q90'][1]\n\n\ndef get_median_estimate(row, quantile_map):\n    x = row['feature_05']\n\n    q10 = quantile_map['q10'][0]\n    q30 = quantile_map['q30'][0]\n    q50 = quantile_map['q50'][0]\n    q70 = quantile_map['q70'][0]\n    q90 = quantile_map['q90'][0]\n    \n    if x<=q10:\n        return quantile_map['q10'][2]\n    elif q10<=x and x<=q30:\n        return quantile_map['q30'][2]\n    elif q30<=x and x<=q50:\n        return quantile_map['q50'][2]\n    elif q50<=x and x<=q70:\n        return quantile_map['q70'][2]\n    else:\n        return quantile_map['q90'][2]\n\ntarget_list = []\npred_mean_list=[]\npred_median_list=[]\nweights=[]\n\nfor i in range(1, len(date_list)):\n    prev_df = symbol_df[symbol_df.date_id==date_list[i-1]]\n    cur_df = symbol_df[symbol_df.date_id==date_list[i]].copy()\n    quantile_map = get_previous_metrics(prev_df)\n\n    cur_df['mean_estimate'] = cur_df.apply(get_mean_estimate, args=(quantile_map, ), axis=1)\n    cur_df['median_estimate'] = cur_df.apply(get_median_estimate, args=(quantile_map, ), axis=1)\n\n    cur_mean_estimate = list(cur_df['mean_estimate'].values)\n    cur_median_estimate = list(cur_df['median_estimate'].values)\n    \n    target_list += list(cur_df['target'].values)\n    pred_mean_list += cur_mean_estimate\n    pred_median_list += cur_median_estimate\n    weights += list(cur_df['weight'].values)\n\ntarget_list = np.array(target_list)\npred_mean_list=np.array(pred_mean_list)\npred_median_list=np.array(pred_median_list)\nweights=np.array(weights)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:24:38.044887Z","iopub.execute_input":"2024-11-13T23:24:38.045454Z","iopub.status.idle":"2024-11-13T23:25:40.627869Z","shell.execute_reply.started":"2024-11-13T23:24:38.045390Z","shell.execute_reply":"2024-11-13T23:25:40.626405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.scatterplot(x=cur_df['feature_05'].values,y=cur_median_estimate)\nsns.scatterplot(x=cur_df['feature_05'].values, y=cur_df['target'].values)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:25:40.629687Z","iopub.execute_input":"2024-11-13T23:25:40.630145Z","iopub.status.idle":"2024-11-13T23:25:40.968714Z","shell.execute_reply.started":"2024-11-13T23:25:40.630097Z","shell.execute_reply":"2024-11-13T23:25:40.967112Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(pred_mean_list[:10])\nprint()\nprint(pred_median_list[:10])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:25:40.971222Z","iopub.execute_input":"2024-11-13T23:25:40.971809Z","iopub.status.idle":"2024-11-13T23:25:40.979627Z","shell.execute_reply.started":"2024-11-13T23:25:40.971746Z","shell.execute_reply":"2024-11-13T23:25:40.978232Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(evaluation_metric(target_list, pred_mean_list, weights))\nprint()\nprint(evaluation_metric(target_list, pred_median_list, weights))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:25:40.981289Z","iopub.execute_input":"2024-11-13T23:25:40.981765Z","iopub.status.idle":"2024-11-13T23:25:41.017552Z","shell.execute_reply.started":"2024-11-13T23:25:40.981713Z","shell.execute_reply":"2024-11-13T23:25:41.016037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"np.corrcoef(pred_mean_list, target_list)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:25:41.019146Z","iopub.execute_input":"2024-11-13T23:25:41.019622Z","iopub.status.idle":"2024-11-13T23:25:41.056054Z","shell.execute_reply.started":"2024-11-13T23:25:41.019557Z","shell.execute_reply":"2024-11-13T23:25:41.054625Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_list = []\npred_list = []\nweights = []\n\ndateid_list = sorted(symbol_df.date_id.unique())\nprint(len(dateid_list))\nfor i in range(10, len(dateid_list)):\n    prev_df = symbol_df[symbol_df.date_id.isin([\n        dateid_list[i-1], \n        dateid_list[i-2],\n        dateid_list[i-3],\n        dateid_list[i-4],\n        dateid_list[i-5],\n        dateid_list[i-6]\n    ])]\n    cur_df = symbol_df[symbol_df.date_id==dateid_list[i]]\n\n    X = prev_df[['feature_05', 'feature_06']].values\n    y = prev_df['target'].values\n\n    model = KNeighborsRegressor(10)\n    #model = Ridge()\n    model.fit(X, y)\n    \n    ypred_prev = model.predict(X)\n    \n    Xnew = cur_df[['feature_05', 'feature_06']].values\n    ynew = cur_df['target'].values\n    ypred = model.predict(Xnew)\n\n    target_list += list(ynew)\n    pred_list += list(ypred)\n    weights += list(cur_df['weight'].values)\ntarget_list = np.array(target_list)\npred_list = np.array(pred_list)\nweights = np.array(weights)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:25:41.058355Z","iopub.execute_input":"2024-11-13T23:25:41.058898Z","iopub.status.idle":"2024-11-13T23:26:35.109509Z","shell.execute_reply.started":"2024-11-13T23:25:41.058832Z","shell.execute_reply":"2024-11-13T23:26:35.108064Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"evaluation_metric(target_list,pred_list, weights)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:26:35.111506Z","iopub.execute_input":"2024-11-13T23:26:35.113132Z","iopub.status.idle":"2024-11-13T23:26:35.132026Z","shell.execute_reply.started":"2024-11-13T23:26:35.113056Z","shell.execute_reply":"2024-11-13T23:26:35.130656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(np.corrcoef(y, ypred_prev)[0, 1])\nprint(np.corrcoef(ynew, ypred)[0, 1])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:26:35.133551Z","iopub.execute_input":"2024-11-13T23:26:35.133970Z","iopub.status.idle":"2024-11-13T23:26:35.142661Z","shell.execute_reply.started":"2024-11-13T23:26:35.133926Z","shell.execute_reply":"2024-11-13T23:26:35.141360Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig,ax = plt.subplots(1, 2, figsize=(15, 5))\n\nsns.scatterplot(x=X[:, 1], y=y, ax=ax[0], label=\"target\")\nsns.scatterplot(x=X[:, 1], y=ypred_prev, ax=ax[0], label=\"predicted\")\n\nsns.scatterplot(x=Xnew[:, 1], y=ynew, ax=ax[1], label=\"target\")\nsns.scatterplot(x=Xnew[:, 1], y=ypred, ax=ax[1], label=\"predicted\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:26:35.144589Z","iopub.execute_input":"2024-11-13T23:26:35.145081Z","iopub.status.idle":"2024-11-13T23:26:35.891251Z","shell.execute_reply.started":"2024-11-13T23:26:35.145029Z","shell.execute_reply":"2024-11-13T23:26:35.889901Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.histplot(ynew, label='target')\nsns.histplot(ypred, label='predicted')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T23:26:35.893695Z","iopub.execute_input":"2024-11-13T23:26:35.894265Z","iopub.status.idle":"2024-11-13T23:26:36.258135Z","shell.execute_reply.started":"2024-11-13T23:26:35.894202Z","shell.execute_reply":"2024-11-13T23:26:36.256737Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**observation**\n1. when using knn, it fits well to the previous day distribution, but not with the current day distribution.\n2. this error in the fit can be attributed to the change in the feature distribution and the target distribution.\n3. change target distribution depends both on the features and the previous day target(for eg. raise in the prices in the previous 2 days may result in the reduction of price at present day). ","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}