{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\n\nfrom collections import defaultdict\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport xgboost as xgb\n\n\nfrom scipy.stats import spearmanr\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.linear_model import Ridge\nfrom sklearn.metrics import r2_score\n\nfrom statsmodels.graphics.tsaplots import plot_acf, plot_pacf\nfrom statsmodels.tsa.stattools import acf\n\nfrom xgboost import XGBRegressor","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-17T13:32:38.556927Z","iopub.execute_input":"2024-11-17T13:32:38.557372Z","iopub.status.idle":"2024-11-17T13:32:42.470643Z","shell.execute_reply.started":"2024-11-17T13:32:38.557324Z","shell.execute_reply":"2024-11-17T13:32:42.469383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SYMBOL_ID=10\nFEATURES = ['feature_06']\n\ntrain_df = []\nfor partition_id in range(10):\n    path = \"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={}/part-0.parquet\".format(partition_id)\n    df = pd.read_parquet(path, \n                         columns=['symbol_id', 'date_id', 'time_id', 'responder_6'] + FEATURES,\n                         filters=[('symbol_id', '==', SYMBOL_ID)]\n                        )\n    \n    #df = df[['symbol_id', 'date_id','time_id', 'responder_6']]\n    df.rename(columns={\n        'responder_6': 'target'\n    }, inplace=True)\n    train_df.append(df)\n\ntrain_df = pd.concat(train_df)\ntrain_df = train_df.sort_values(['symbol_id', 'date_id', 'time_id'])\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T13:32:42.475371Z","iopub.execute_input":"2024-11-17T13:32:42.475941Z","iopub.status.idle":"2024-11-17T13:32:49.097550Z","shell.execute_reply.started":"2024-11-17T13:32:42.475896Z","shell.execute_reply":"2024-11-17T13:32:49.096321Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**In this notebook, we explore the relation between the feature variable and the target variable.**","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(13,5))\nsns.heatmap(train_df[['feature_06', 'target']].corr(), annot=True, fmt='.3f', ax=ax[0])\nsns.scatterplot(data=train_df, x='feature_06', y='target', ax=ax[1])\n\nax[0].xaxis.tick_top()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T13:32:49.099085Z","iopub.execute_input":"2024-11-17T13:32:49.099622Z","iopub.status.idle":"2024-11-17T13:32:52.766336Z","shell.execute_reply.started":"2024-11-17T13:32:49.099568Z","shell.execute_reply":"2024-11-17T13:32:52.764960Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.scatterplot(\n    data=train_df[train_df.date_id==13],\n    x='feature_06',\n    y='target'\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T13:32:52.768997Z","iopub.execute_input":"2024-11-17T13:32:52.769394Z","iopub.status.idle":"2024-11-17T13:32:53.107317Z","shell.execute_reply.started":"2024-11-17T13:32:52.769354Z","shell.execute_reply":"2024-11-17T13:32:53.105940Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df[train_df.date_id==100][['feature_06', 'target']].corr()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T13:32:53.108649Z","iopub.execute_input":"2024-11-17T13:32:53.109032Z","iopub.status.idle":"2024-11-17T13:32:53.123513Z","shell.execute_reply.started":"2024-11-17T13:32:53.108992Z","shell.execute_reply":"2024-11-17T13:32:53.122311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dateid_list = sorted(train_df['date_id'].unique())\ncorr_list = []\nfor it, date_id in enumerate(dateid_list):\n    df = train_df[train_df.date_id==date_id]\n    corr = np.corrcoef(df['feature_06'], df['target'])[0, 1]\n    corr_list.append(corr)\n\n\nfig, ax = plt.subplots(1, 2, figsize=(14, 4))\nfig.suptitle(\"distribution of correlations between feature and target\")\nsns.histplot(corr_list, ax=ax[0])\nsns.lineplot(x=np.arange(len(dateid_list)), y=corr_list, ax=ax[1])\nax[1].set_xlabel(\"date_id\")\nax[1].set_ylabel(\"correlation\")\nplt.show()\n\n\n\n\nrolling_corr_list = []\ncorr_list = np.array(corr_list)\nfor i in range(10, len(corr_list)):\n    rolling_corr_list.append(corr_list[i-10: i].mean())\n\n\n\nfig, ax = plt.subplots(1, 3, figsize=(14, 4))\nfig.suptitle(\"correlation over time\")\nax[0].plot(corr_list[-100:])\nax[1].plot(np.cumsum(corr_list)/np.arange(1, 1+len(corr_list)))\nax[2].plot(rolling_corr_list)\n\nplt.xlabel(\"date_id\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T13:32:53.124948Z","iopub.execute_input":"2024-11-17T13:32:53.125335Z","iopub.status.idle":"2024-11-17T13:32:56.388026Z","shell.execute_reply.started":"2024-11-17T13:32:53.125296Z","shell.execute_reply":"2024-11-17T13:32:56.386852Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Observations**\n1. Overall Correlation between feature and target is around -0.06\n2. Correlation varies per day.\n3. Correlation during the initial dates are more negative than later periods. ","metadata":{}},{"cell_type":"code","source":"fig,ax=plt.subplots(1, 2, figsize=(13,4))\nplot_acf(np.array(corr_list), ax=ax[0], title=\"Correlation ACF\")\nplot_pacf(np.array(corr_list), ax=ax[1], title=\"Correlation PACF\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T13:32:56.389610Z","iopub.execute_input":"2024-11-17T13:32:56.390030Z","iopub.status.idle":"2024-11-17T13:32:56.893517Z","shell.execute_reply.started":"2024-11-17T13:32:56.389984Z","shell.execute_reply":"2024-11-17T13:32:56.892446Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"> We have seen the point-wise correlations between the feature and target changes with time.\n> Point-Wise Correlation even exists is not highly linear per day\n> We will continue to analysis the correlations and change of distributions for both feature and target variable ","metadata":{}},{"cell_type":"markdown","source":"#### quantile stats for the feature and target","metadata":{}},{"cell_type":"code","source":"stat_df = train_df[['symbol_id', 'date_id']].drop_duplicates()\nfor q in [0.1,0.3, 0.5, 0.7, 0.9]:\n    df = train_df.groupby(['symbol_id', 'date_id'])[['target','feature_06']].quantile(q).rename(columns={'target': 'target_'+str(q), 'feature_06': 'feature_06_'+str(q)}).reset_index()\n    stat_df = stat_df.merge(df)\nstat_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T13:32:56.895239Z","iopub.execute_input":"2024-11-17T13:32:56.895576Z","iopub.status.idle":"2024-11-17T13:32:58.307858Z","shell.execute_reply.started":"2024-11-17T13:32:56.895541Z","shell.execute_reply":"2024-11-17T13:32:58.306801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig,ax=plt.subplots(1,2, figsize=(15,4))\n\nTARGET_COLUMNS = [colname for colname in stat_df.columns if colname.startswith('target_')]\nFEATURE_COLUMNS = [colname for colname in stat_df.columns if colname.startswith('feature_')]\n\nsns.heatmap(    \n    stat_df[TARGET_COLUMNS].corr(),\n    annot=True,\n    fmt='.2f',\n    ax=ax[0],\n    cbar=False\n)\nsns.heatmap(    \n    stat_df[FEATURE_COLUMNS].corr(),\n    annot=True,\n    fmt='.2f',\n    ax=ax[1],\n    cbar=False\n)\nax[0].xaxis.tick_top()\nax[1].xaxis.tick_top()\n\nax[1].xaxis.set_tick_params(rotation=45)\nax[1].yaxis.set_tick_params(rotation=45)\nax[0].xaxis.set_tick_params(rotation=45)\nax[0].yaxis.set_tick_params(rotation=45)\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T13:32:58.309363Z","iopub.execute_input":"2024-11-17T13:32:58.309825Z","iopub.status.idle":"2024-11-17T13:32:58.868539Z","shell.execute_reply.started":"2024-11-17T13:32:58.309771Z","shell.execute_reply":"2024-11-17T13:32:58.867397Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(14, 5))\n\nSTAT_FEATURES = [colname for colname in stat_df.columns if colname.startswith('feature_')]\nSTAT_TARGET = [colname for colname in stat_df.columns if colname.startswith('target_')]\n\nax = sns.heatmap(\n    stat_df[STAT_FEATURES+STAT_TARGET].corr(),\n    annot=True,\n    fmt='.2f',\n    cbar=False\n)\n\nax.xaxis.set_tick_params(rotation=45)\nax.xaxis.tick_top()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T13:32:58.872691Z","iopub.execute_input":"2024-11-17T13:32:58.874228Z","iopub.status.idle":"2024-11-17T13:32:59.398165Z","shell.execute_reply.started":"2024-11-17T13:32:58.874164Z","shell.execute_reply":"2024-11-17T13:32:59.396918Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Observations**\n1. Show above are the overall correlations between the quantiles of target and feature variables as per day basis.\n2. Correlation increases between various quantiles of featuter and target variables.\n3. Correlation also increases between the target and the feature variable quantiles","metadata":{}},{"cell_type":"markdown","source":"Lets us see some timeplots to understand the dynamics of the variables","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(14, 4))\nsns.lineplot(\n    data=stat_df.head(200),\n    x='date_id',\n    y='feature_06_0.1',\n    label='feature_06_0.1'\n)\n\nsns.lineplot(\n    data=stat_df.head(200),\n    x='date_id',\n    y='feature_06_0.9',\n    label='feature_06_0.9'\n)\n\nsns.lineplot(\n    data=stat_df.head(200),\n    x='date_id',\n    y='target_0.1',\n    label='target_0.1'\n)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T13:32:59.399656Z","iopub.execute_input":"2024-11-17T13:32:59.400116Z","iopub.status.idle":"2024-11-17T13:32:59.766294Z","shell.execute_reply.started":"2024-11-17T13:32:59.400062Z","shell.execute_reply":"2024-11-17T13:32:59.765149Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"ACF and PACF plots for the feature and target variables","metadata":{}},{"cell_type":"code","source":"for colname in FEATURE_COLUMNS + TARGET_COLUMNS:\n    fig,ax = plt.subplots(1, 2, figsize=(14,4))\n    fig.suptitle(colname)\n    plot_acf(stat_df[colname], ax=ax[0])\n    plot_pacf(stat_df[colname], ax=ax[1])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T13:32:59.768397Z","iopub.execute_input":"2024-11-17T13:32:59.768869Z","iopub.status.idle":"2024-11-17T13:33:05.676567Z","shell.execute_reply.started":"2024-11-17T13:32:59.768794Z","shell.execute_reply":"2024-11-17T13:33:05.675305Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for feat_col in FEATURE_COLUMNS:\n    for target_col in TARGET_COLUMNS:\n        print(feat_col, target_col)\n        xfeat = stat_df[feat_col].values\n        xtarget = stat_df[target_col].values\n\n        WINDOW=20\n        win_corr_list = []\n        for i in range(WINDOW, len(xfeat)):\n            corr = np.corrcoef(xfeat[i-WINDOW: i], xtarget[i-WINDOW: i])[0, 1]\n            win_corr_list.append(corr)\n\n        exp_mv_corr_list= np.zeros(len(win_corr_list))\n        exp_mv_corr_list[0] = win_corr_list[0]\n        for i in range(1, len(win_corr_list)):\n            exp_mv_corr_list[i] = 0.9*win_corr_list[i] + 0.1 * exp_mv_corr_list[i-1]\n\n        fig,ax=plt.subplots(1, 3, figsize=(15, 5))\n        fig.suptitle(feat_col + \" | \"+target_col)\n        sns.histplot(x=win_corr_list, ax=ax[0])\n        ax[1].plot(exp_mv_corr_list[300:800])\n        ax[2].plot( (np.cumsum(win_corr_list)/np.arange(1, 1+len(win_corr_list)))[1:] )\n        \n        plt.show()\n    break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T13:33:05.678495Z","iopub.execute_input":"2024-11-17T13:33:05.678894Z","iopub.status.idle":"2024-11-17T13:33:09.208563Z","shell.execute_reply.started":"2024-11-17T13:33:05.678853Z","shell.execute_reply":"2024-11-17T13:33:09.207325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stat_df[stat_df.date_id<500][FEATURE_COLUMNS + TARGET_COLUMNS].corr()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T13:33:09.210124Z","iopub.execute_input":"2024-11-17T13:33:09.210547Z","iopub.status.idle":"2024-11-17T13:33:09.234698Z","shell.execute_reply.started":"2024-11-17T13:33:09.210501Z","shell.execute_reply":"2024-11-17T13:33:09.233572Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Observations**\n1. for both feature and target variable, LAG correlation for median is very low.\n2. for the rest of the quantiles, it appears to have better correlation.\n3. Even in this case of estimating the quantiles, correlations between the target and feature is changing with time, but in this case with higher correlation.\n4. Density zero correlation is  very less when estimating the quantiles.","metadata":{}},{"cell_type":"markdown","source":"**Summary of observations so far...**\n\nPoint-wise Analysis\n1. Distribution of target and feature changes for every date.\n2. correlation between target and feature(linear relation) changes every date.\n3. Correlation around zero have higher density.\n4. We observe negative correlation intially dominated followed by positive correlation(this could also be due to the othe factors)\n\nQuantile-Wise Analysis\n1. Overall Strong correlation is observed between the feature and the target.\n2. Density of correlations around zero is less and skewed to more positive and negative.\n3. Linear correlation between quantile target and feature_06 has dynamic relation with time(changes with time).\n4. AR-models can be used to model timeseries.","metadata":{}},{"cell_type":"markdown","source":"**Target Modelling at a quantile**\n\ntarget\n1. AR(3) on the target - target_ar\n2. AR(3) on the feature - feature_ar\n3. (target_ar, feature_ar) - target_pred","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"from statsmodels.tsa.arima.model import ARIMA\nfrom sklearn.metrics import mean_absolute_percentage_error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T13:33:09.236420Z","iopub.execute_input":"2024-11-17T13:33:09.236855Z","iopub.status.idle":"2024-11-17T13:33:09.530545Z","shell.execute_reply.started":"2024-11-17T13:33:09.236797Z","shell.execute_reply":"2024-11-17T13:33:09.529103Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Global AR(3) models for target and feature variables","metadata":{}},{"cell_type":"code","source":"stat_df = stat_df.sort_values('date_id')\ndateid_list = sorted(stat_df['date_id'].unique())\n\ntrain_dateid_list = dateid_list[:-100]\nval_dateid_list = dateid_list[-100:]\n\nprint(len(train_dateid_list), len(val_dateid_list))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T13:33:09.532930Z","iopub.execute_input":"2024-11-17T13:33:09.533675Z","iopub.status.idle":"2024-11-17T13:33:09.543630Z","shell.execute_reply.started":"2024-11-17T13:33:09.533618Z","shell.execute_reply":"2024-11-17T13:33:09.542308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def model_arma(y, mp, colname, lag=3):\n    arma_mod = ARIMA(y, order=(lag, 0, 0)).fit()\n    yhat = arma_mod.predict()\n    return arma_mod","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T14:14:22.048001Z","iopub.execute_input":"2024-11-17T14:14:22.048446Z","iopub.status.idle":"2024-11-17T14:14:22.055471Z","shell.execute_reply.started":"2024-11-17T14:14:22.048403Z","shell.execute_reply":"2024-11-17T14:14:22.054123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_ar_models = {}\ntarget_ar_models = {}\nfor colname in FEATURE_COLUMNS + TARGET_COLUMNS:\n    y = stat_df[stat_df.date_id.isin(train_dateid_list)][colname].values\n    if colname.startswith('feature'):\n        feature_ar_models[colname] = model_arma(y, feature_ar_models, colname)\n    elif colname.startswith('target'):\n        target_ar_models[colname] = model_arma(y, feature_ar_models, colname)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T13:33:09.555982Z","iopub.execute_input":"2024-11-17T13:33:09.556391Z","iopub.status.idle":"2024-11-17T13:33:15.451876Z","shell.execute_reply.started":"2024-11-17T13:33:09.556351Z","shell.execute_reply":"2024-11-17T13:33:15.450730Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"metric_df=[]\nfor COLNAME in FEATURE_COLUMNS + TARGET_COLUMNS:\n    model=None\n    if COLNAME.startswith('feature'):\n        model = feature_ar_models[COLNAME]\n    elif COLNAME.startswith('target'):\n        model = target_ar_models[COLNAME]\n\n    ytrain = stat_df[stat_df.date_id.isin(train_dateid_list)][COLNAME].values\n    yval = stat_df[stat_df.date_id.isin(val_dateid_list)][COLNAME].values\n    res = model\n    ytrain_hat = res.predict()\n    res = res.append(yval, refit=False)\n    \n    yval_hat = res.predict()[len(ytrain): ]\n\n    metric_df.append({\n        'colname': COLNAME,\n        'mode': 'train',\n        'mape': mean_absolute_percentage_error(ytrain, ytrain_hat)\n    })\n\n    metric_df.append({\n        'colname': COLNAME,\n        'mode': 'eval',\n        'mape': mean_absolute_percentage_error(yval, yval_hat)\n    })\n    \n    print(\"train MAPE:{:.2f}\".format(mean_absolute_percentage_error(ytrain, ytrain_hat)))\n    print(\"val MAPE:{:.2f}\".format(mean_absolute_percentage_error(yval, yval_hat)))\n    \n    fig, ax = plt.subplots(1, 2, figsize=(14,4))\n    fig.suptitle(COLNAME)\n    ax[0].plot(ytrain, label='train-true')\n    ax[0].plot(ytrain_hat, label='train-pred')\n    \n    ax[1].plot(yval, label='val-true')\n    ax[1].plot(yval_hat, label='val-pred')\n    plt.legend()\n    plt.show()\n\nmetric_df = pd.DataFrame.from_dict(metric_df)\nmetric_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T13:33:15.458873Z","iopub.execute_input":"2024-11-17T13:33:15.462119Z","iopub.status.idle":"2024-11-17T13:33:20.318322Z","shell.execute_reply.started":"2024-11-17T13:33:15.462049Z","shell.execute_reply":"2024-11-17T13:33:20.317049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"metric_df.sort_values(['mode', 'colname'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T13:33:20.319979Z","iopub.execute_input":"2024-11-17T13:33:20.320484Z","iopub.status.idle":"2024-11-17T13:33:20.337667Z","shell.execute_reply.started":"2024-11-17T13:33:20.320431Z","shell.execute_reply":"2024-11-17T13:33:20.335857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(13, 5))\nplt.title(\"AR(3) model metrics:MAPE\")\nsns.barplot(data=metric_df[metric_df['colname'].isin(['feature_06_0.5', 'target_0.5'])==False], x='colname', y='mape', hue='mode')\nplt.xticks(rotation=45)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T13:33:20.339214Z","iopub.execute_input":"2024-11-17T13:33:20.339728Z","iopub.status.idle":"2024-11-17T13:33:20.957752Z","shell.execute_reply.started":"2024-11-17T13:33:20.339676Z","shell.execute_reply":"2024-11-17T13:33:20.956362Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The above models share parameters at all the timesteps.\n\nLets check if model can forecast 1-step ahead, based on recent past data(say 15 days)","metadata":{}},{"cell_type":"code","source":"metric_df=[]\nfor COLNAME in FEATURE_COLUMNS + TARGET_COLUMNS:\n    y = stat_df[COLNAME].values[-150:]\n    WINDOW = 50\n\n    ytrue = []\n    ypred = []\n    \n    for i in range(WINDOW, len(y)-1):\n        ywin = y[i-WINDOW: i+1]\n        res = model_arma(ywin, feature_ar_models, COLNAME)\n        ytrue.append(y[i+1])\n        ypred.append(res.forecast(1))\n\n    yval = ytrue[-100:]\n    yval_hat = ypred[-100:]\n\n    plt.title(COLNAME)\n    plt.plot(yval, label=\"val-true\")\n    plt.plot(yval_hat, label=\"val-pred\")\n    plt.show()\n    metric_df.append({\n        'colname': COLNAME,\n        'mode': 'eval',\n        'mape': mean_absolute_percentage_error(yval, yval_hat)\n    })\nmetric_df = pd.DataFrame.from_dict(metric_df)\nmetric_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T14:25:42.125797Z","iopub.execute_input":"2024-11-17T14:25:42.126319Z","iopub.status.idle":"2024-11-17T14:27:07.528384Z","shell.execute_reply.started":"2024-11-17T14:25:42.126273Z","shell.execute_reply":"2024-11-17T14:27:07.527082Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"metric_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T14:27:19.871948Z","iopub.execute_input":"2024-11-17T14:27:19.872445Z","iopub.status.idle":"2024-11-17T14:27:19.885709Z","shell.execute_reply.started":"2024-11-17T14:27:19.872403Z","shell.execute_reply":"2024-11-17T14:27:19.884501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}