{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.11.11"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# DRW - Crypto Market Prediction | EDA","metadata":{}},{"cell_type":"markdown","source":"## Interpreting Results\n\n--- Work in progress ---","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport datetime as dt\n\nimport matplotlib.dates as md\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom matplotlib import cycler\nfrom matplotlib.gridspec import GridSpec\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\ncolors = [\"#53599A\", \"#068D9D\", \"#607BB0\", \"#77BECF\", \"#6D9DC5\", \"#80DED9\", \"#AEECEF\"]\nplt.rc('axes', facecolor=\"#E9E9E9\", edgecolor='none', axisbelow=True, grid=True, prop_cycle=cycler('color', colors))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T15:48:41.946033Z","iopub.execute_input":"2025-06-12T15:48:41.946312Z","iopub.status.idle":"2025-06-12T15:48:43.360615Z","shell.execute_reply.started":"2025-06-12T15:48:41.946281Z","shell.execute_reply":"2025-06-12T15:48:43.359598Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Helpful functions","metadata":{}},{"cell_type":"code","source":"def reduce_mem_usage(dataframe):\n    initial_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    for col in dataframe.columns:\n        col_type = dataframe[col].dtype\n\n        c_min = dataframe[col].min()\n        c_max = dataframe[col].max()\n        if str(col_type)[:3] == 'int':\n            if c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                dataframe[col] = dataframe[col].astype(np.int32)\n            else:\n                dataframe[col] = dataframe[col].astype(np.int64)\n        else:\n            if c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                dataframe[col] = dataframe[col].astype(np.float32)\n            else:\n                dataframe[col] = dataframe[col].astype(np.float64)\n\n    final_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    print('--- Memory usage before: {:.2f} MB'.format(initial_mem_usage))\n    print('--- Memory usage after: {:.2f} MB'.format(final_mem_usage))\n    print('--- Decreased memory usage by {:.1f}%\\n'.format(100 * (initial_mem_usage - final_mem_usage) / initial_mem_usage))\n\n    return dataframe\n\n\ndef plot_relationship(pairs, plot_df, cols=5):\n    rows = np.ceil(len(pairs)/cols).astype(int)\n    fig, axs = plt.subplots(nrows=rows, ncols=cols, figsize=(20, 4*rows))\n    for pair in pairs:\n        ax = axs[pairs.index(pair) // cols, pairs.index(pair) % cols]\n        ax.scatter(plot_df[pair[0]], plot_df[pair[1]], s=0.2)\n        ax.set_title(f\"{pair[0]} vs {pair[1]}\\nCorrelation: {pair[2]:.2f}\")\n        beta, alpha = np.polyfit(plot_df[pair[0]], plot_df[pair[1]], deg=1)\n        ax.axline(xy1=(0, alpha), slope=1, color='red', linestyle='--')\n        \n    plt.tight_layout()\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T15:48:43.362860Z","iopub.execute_input":"2025-06-12T15:48:43.363322Z","iopub.status.idle":"2025-06-12T15:48:43.375171Z","shell.execute_reply.started":"2025-06-12T15:48:43.363297Z","shell.execute_reply":"2025-06-12T15:48:43.373620Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Loading data","metadata":{}},{"cell_type":"code","source":"data_dir = '/kaggle/input/drw-crypto-market-prediction/'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T15:48:43.376566Z","iopub.execute_input":"2025-06-12T15:48:43.376944Z","iopub.status.idle":"2025-06-12T15:48:43.405943Z","shell.execute_reply.started":"2025-06-12T15:48:43.376892Z","shell.execute_reply":"2025-06-12T15:48:43.405195Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_parquet(os.path.join(data_dir, 'train.parquet'))\ndf_test = pd.read_parquet(os.path.join(data_dir, 'test.parquet'))\n\ndf_train = reduce_mem_usage(df_train)\ndf_test = reduce_mem_usage(df_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T15:48:43.407015Z","iopub.execute_input":"2025-06-12T15:48:43.407321Z","iopub.status.idle":"2025-06-12T15:49:49.216197Z","shell.execute_reply.started":"2025-06-12T15:48:43.407292Z","shell.execute_reply":"2025-06-12T15:49:49.214729Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('Train shape:', df_train.shape)\nprint('Test shape:', df_test.shape)\nprint(\"df_train memory usage: {:.2f} GB\".format(df_train.memory_usage(deep=True).sum() / 1024**3))\nprint(\"df_test memory usage: {:.2f} GB\".format(df_test.memory_usage(deep=True).sum() / 1024**3))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T15:49:49.218839Z","iopub.execute_input":"2025-06-12T15:49:49.219771Z","iopub.status.idle":"2025-06-12T15:49:49.249383Z","shell.execute_reply.started":"2025-06-12T15:49:49.219730Z","shell.execute_reply":"2025-06-12T15:49:49.248175Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Quality Checks","metadata":{}},{"cell_type":"code","source":"# Check for Inf\ninf_idx = (abs(df_train)==np.inf).transpose().sum()\ninf_idx = inf_idx[inf_idx!=0]\nprint(f\"Found {len(inf_idx)}\\t observations with Inf values.\")\n\ninf_idx = (abs(df_train)==np.inf).sum()\ninf_idx = inf_idx[inf_idx!=0]\nprint(f\"Found {len(inf_idx)}\\t columns with Inf values.\")\n\nprint(\"Removing columns\")\ndf_train = df_train.drop(columns=inf_idx.index)\ndf_test = df_test.drop(columns=inf_idx.index)\n\nprint(\"New Shapes:\")\nprint(\"Training Data:\\t\", df_train.shape)\nprint(\"Test Data:\\t\", df_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T15:49:49.250669Z","iopub.execute_input":"2025-06-12T15:49:49.251022Z","iopub.status.idle":"2025-06-12T15:50:11.395433Z","shell.execute_reply.started":"2025-06-12T15:49:49.250970Z","shell.execute_reply":"2025-06-12T15:50:11.394143Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check for constant features\nconst_feat = df_train.var()==0\nconst_feat = const_feat[const_feat]\nprint(f\"Found {len(const_feat)} constant features.\")\nprint(\"Removing columns\")\ndf_train = df_train[ list(set(df_train.columns) - set(const_feat.index)) ]\ndf_test = df_test[ list(set(df_test.columns) - set(const_feat.index)) ]\n\nprint(\"New Shapes:\")\nprint(\"Training Data:\\t\", df_train.shape)\nprint(\"Test Data:\\t\", df_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T15:50:11.398584Z","iopub.execute_input":"2025-06-12T15:50:11.398885Z","iopub.status.idle":"2025-06-12T15:50:16.946326Z","shell.execute_reply.started":"2025-06-12T15:50:11.398862Z","shell.execute_reply":"2025-06-12T15:50:16.944131Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Deep Dive","metadata":{}},{"cell_type":"code","source":"train_x = df_train.drop(columns=['label'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T15:50:16.947793Z","iopub.execute_input":"2025-06-12T15:50:16.948154Z","iopub.status.idle":"2025-06-12T15:50:18.153401Z","shell.execute_reply.started":"2025-06-12T15:50:16.948125Z","shell.execute_reply":"2025-06-12T15:50:18.152306Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Outlier Detection","metadata":{}},{"cell_type":"code","source":"n_cols = train_x.shape[1]\noutlier_threshold = 0.2 * n_cols\nprint(f\"Outlier threshold: {outlier_threshold/n_cols:.1%}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T15:50:18.154577Z","iopub.execute_input":"2025-06-12T15:50:18.154911Z","iopub.status.idle":"2025-06-12T15:50:18.161165Z","shell.execute_reply.started":"2025-06-12T15:50:18.154881Z","shell.execute_reply":"2025-06-12T15:50:18.159924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"z_score_df = (train_x - train_x.mean()) / train_x.std()\npotential_outliers = z_score_df[(np.abs(z_score_df) > 3).sum(axis=1) > outlier_threshold].index\nprint(f\"Found {len(potential_outliers)} potential outliers based on Z-score method.\")\noutliers = potential_outliers","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T15:50:18.162197Z","iopub.execute_input":"2025-06-12T15:50:18.162498Z","iopub.status.idle":"2025-06-12T15:50:32.877385Z","shell.execute_reply.started":"2025-06-12T15:50:18.162475Z","shell.execute_reply":"2025-06-12T15:50:32.876398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"z_score_df = 0.6745 * (train_x - train_x.median()) / (train_x - train_x.median()).abs().median()\npotential_outliers = z_score_df[(np.abs(z_score_df) > 3).sum(axis=1) > outlier_threshold].index\nprint(f\"Found {len(potential_outliers)} potential outliers based on modified Z-score method.\")\npotential_outliers = outliers.intersection(potential_outliers)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T15:50:32.878318Z","iopub.execute_input":"2025-06-12T15:50:32.878621Z","iopub.status.idle":"2025-06-12T15:50:54.828709Z","shell.execute_reply.started":"2025-06-12T15:50:32.878582Z","shell.execute_reply":"2025-06-12T15:50:54.827182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"q25 = train_x.quantile(0.25, axis=0)\nq75 = train_x.quantile(0.75, axis=0)\niqr = q75 - q25\npotential_outliers = train_x[((train_x < q25-1.5*iqr) | (train_x > q75+1.5*iqr)).sum(axis=1) > outlier_threshold].index\nprint(f\"Found {len(potential_outliers)} potential outliers based on IQR method.\")\npotential_outliers = outliers.intersection(potential_outliers)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T15:50:54.829439Z","iopub.execute_input":"2025-06-12T15:50:54.829696Z","iopub.status.idle":"2025-06-12T15:51:12.016170Z","shell.execute_reply.started":"2025-06-12T15:50:54.829676Z","shell.execute_reply":"2025-06-12T15:51:12.014875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop outliers\ndf_train = df_train.drop(index=potential_outliers)\nprint(f\"Dropping {len(potential_outliers)} outliers from training data.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T15:51:12.017136Z","iopub.execute_input":"2025-06-12T15:51:12.017393Z","iopub.status.idle":"2025-06-12T15:51:15.509939Z","shell.execute_reply.started":"2025-06-12T15:51:12.017372Z","shell.execute_reply":"2025-06-12T15:51:15.509034Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Market Data Distributions","metadata":{}},{"cell_type":"code","source":"train_x_mkt = train_x[[col for col in train_x.columns if 'X' not in col]].sort_index(axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T15:51:15.510923Z","iopub.execute_input":"2025-06-12T15:51:15.511213Z","iopub.status.idle":"2025-06-12T15:51:15.533846Z","shell.execute_reply.started":"2025-06-12T15:51:15.511191Z","shell.execute_reply":"2025-06-12T15:51:15.532516Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Single Day Analysis","metadata":{}},{"cell_type":"code","source":"date = dt.date(2024,1,20)\nplot_df = train_x_mkt.loc[train_x_mkt.index.date==date].copy()\nfig, axs = plt.subplots(nrows=plot_df.shape[1], ncols=2, figsize=(12, 6), gridspec_kw={'width_ratios': [3, 1]})\nfig.suptitle(f'Market Features on {date}', fontsize=16)\nfor i, col in enumerate(plot_df.columns):\n    axs[i,0].scatter(plot_df.index, plot_df[col], label=col, color=colors[i], s=0.2)\n    axs[i,0].legend(loc='upper right', fontsize='small')\n    axs[i,0].xaxis.set_major_formatter(md.DateFormatter('%H:%M'))\n    axs[i,1].hist(plot_df[col][plot_df[col].between(0, plot_df[col].quantile(0.95))], color=colors[i], bins=50)\n    \nfig.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T15:51:15.535182Z","iopub.execute_input":"2025-06-12T15:51:15.535509Z","iopub.status.idle":"2025-06-12T15:51:17.907216Z","shell.execute_reply.started":"2025-06-12T15:51:15.535476Z","shell.execute_reply":"2025-06-12T15:51:17.905939Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.heatmap(train_x_mkt.corr(), annot=True, fmt='.2f', cmap='coolwarm', vmin=-1, vmax=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T15:51:17.908374Z","iopub.execute_input":"2025-06-12T15:51:17.908671Z","iopub.status.idle":"2025-06-12T15:51:18.332215Z","shell.execute_reply.started":"2025-06-12T15:51:17.908647Z","shell.execute_reply":"2025-06-12T15:51:18.331006Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Daily Patterns","metadata":{}},{"cell_type":"code","source":"train_x_mkt_d = train_x_mkt.groupby(train_x_mkt.index.date).aggregate(['mean', 'std', 'min', 'max', 'sum'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T15:51:18.333515Z","iopub.execute_input":"2025-06-12T15:51:18.333856Z","iopub.status.idle":"2025-06-12T15:51:18.668234Z","shell.execute_reply.started":"2025-06-12T15:51:18.333825Z","shell.execute_reply":"2025-06-12T15:51:18.667074Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"width = 1\nfig, axs = plt.subplots(1, 1, figsize=(14, 4))\nfig.suptitle('Daily Trading Volume', fontsize=16)\naxs.bar(train_x_mkt_d.index, train_x_mkt_d[('buy_qty', 'sum')], color='green', width=width, label='Bought Quantity')\naxs.bar(train_x_mkt_d.index, train_x_mkt_d[('sell_qty', 'sum')], color='firebrick', width=width, label='Sold Quantity', bottom=train_x_mkt_d[('buy_qty', 'sum')])\naxs.legend(loc='upper right', fontsize='small')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T15:51:18.669416Z","iopub.execute_input":"2025-06-12T15:51:18.669830Z","iopub.status.idle":"2025-06-12T15:51:19.946880Z","shell.execute_reply.started":"2025-06-12T15:51:18.669742Z","shell.execute_reply":"2025-06-12T15:51:19.945881Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Anonymised Data","metadata":{}},{"cell_type":"code","source":"train_x_X = train_x[[col for col in train_x.columns if 'X' in col]].sort_index(axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T15:51:19.947794Z","iopub.execute_input":"2025-06-12T15:51:19.948061Z","iopub.status.idle":"2025-06-12T15:51:23.746711Z","shell.execute_reply.started":"2025-06-12T15:51:19.948042Z","shell.execute_reply":"2025-06-12T15:51:23.745802Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Correlations","metadata":{}},{"cell_type":"code","source":"cor_mat_raw = train_x_X.corr()\ncor_mat = cor_mat_raw.copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T15:51:23.747683Z","iopub.execute_input":"2025-06-12T15:51:23.747938Z","iopub.status.idle":"2025-06-12T16:09:50.749226Z","shell.execute_reply.started":"2025-06-12T15:51:23.747917Z","shell.execute_reply":"2025-06-12T16:09:50.748100Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(8,6))\nsns.heatmap(cor_mat_raw, annot=False, fmt='.2f', cmap='coolwarm', vmin=-1, vmax=1, ax=ax)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T16:09:50.750562Z","iopub.execute_input":"2025-06-12T16:09:50.751056Z","iopub.status.idle":"2025-06-12T16:09:52.640064Z","shell.execute_reply.started":"2025-06-12T16:09:50.751020Z","shell.execute_reply":"2025-06-12T16:09:52.639092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cor_mat = pd.DataFrame(np.triu(cor_mat, k=1), index=cor_mat.index, columns=cor_mat.columns)\n\nperfect_pairs = []\nfor row in cor_mat.iterrows():\n    for col in row[1].index:\n        if abs(row[1][col])==1:\n            perfect_pairs.append((row[0], col, row[1][col]))\n            \nprint(f\"Found {len(perfect_pairs)} perfect pairs of features.\")\nremove = []\nfor pair in perfect_pairs:\n    if pair[0] not in remove:\n        remove.append(pair[0])\n    else:\n        remove.append(pair[1])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T16:09:52.645544Z","iopub.execute_input":"2025-06-12T16:09:52.645897Z","iopub.status.idle":"2025-06-12T16:09:54.350893Z","shell.execute_reply.started":"2025-06-12T16:09:52.645870Z","shell.execute_reply":"2025-06-12T16:09:54.349861Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_relationship(perfect_pairs[:15], train_x_X, cols=5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T16:09:54.351776Z","iopub.execute_input":"2025-06-12T16:09:54.352294Z","iopub.status.idle":"2025-06-12T16:10:05.959619Z","shell.execute_reply.started":"2025-06-12T16:09:54.352259Z","shell.execute_reply":"2025-06-12T16:10:05.958668Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"uncorrelated_pairs = []\nfor row in cor_mat.iterrows():\n    for col in row[1].index:\n        if row[0]!=col and abs(row[1][col])<0.01:\n            uncorrelated_pairs.append((row[0], col, row[1][col]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T16:10:05.960945Z","iopub.execute_input":"2025-06-12T16:10:05.961330Z","iopub.status.idle":"2025-06-12T16:10:09.811912Z","shell.execute_reply.started":"2025-06-12T16:10:05.961303Z","shell.execute_reply":"2025-06-12T16:10:09.810900Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_relationship(uncorrelated_pairs[:15], train_x_X, cols=5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T16:10:09.813049Z","iopub.execute_input":"2025-06-12T16:10:09.813298Z","iopub.status.idle":"2025-06-12T16:10:16.494316Z","shell.execute_reply.started":"2025-06-12T16:10:09.813279Z","shell.execute_reply":"2025-06-12T16:10:16.492961Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Distributions","metadata":{}},{"cell_type":"code","source":"# rows = int(np.ceil(train_x_X.shape[1]/80))\n# fig, axs = plt.subplots(rows, 1, figsize=(14,3*rows))\n\n# for i in range(rows):\n#     plot_df = train_x_X.iloc[:, (i*80):((i+1)*80)]\n\n#     axs[i].violinplot(plot_df, showmeans=False, showmedians=True)\n#     axs[i].xaxis.grid(False)\n#     axs[i].set_xticks([y + 1 for y in range(plot_df.shape[1])],\n#                     labels=plot_df.columns, rotation=90)\n# plt.tight_layout()\n# plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T16:10:16.495659Z","iopub.execute_input":"2025-06-12T16:10:16.497823Z","iopub.status.idle":"2025-06-12T16:10:16.503046Z","shell.execute_reply.started":"2025-06-12T16:10:16.497778Z","shell.execute_reply":"2025-06-12T16:10:16.501550Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Correlations with target","metadata":{}},{"cell_type":"code","source":"target_correlation = {}\ntarget = df_train['label']\nfor col in df_train.drop(columns='label'):\n    target_correlation[col] = np.corrcoef(df_train[col], target)[0][1]\ntarget_correlation = pd.DataFrame(target_correlation.values(), index=target_correlation.keys(), columns=['Corr']).sort_values(by='Corr')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T16:10:16.504342Z","iopub.execute_input":"2025-06-12T16:10:16.504771Z","iopub.status.idle":"2025-06-12T16:10:26.659588Z","shell.execute_reply.started":"2025-06-12T16:10:16.504742Z","shell.execute_reply":"2025-06-12T16:10:26.658696Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axs = plt.subplots(1, 1, figsize=(12,4))\naxs.bar(range(len(target_correlation)), target_correlation['Corr'])\naxs.set_xlabel('Feature')\naxs.set_ylabel('Pearson Correlation')\naxs.set_title('Feature vs Target Correlation')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T16:10:26.660564Z","iopub.execute_input":"2025-06-12T16:10:26.660823Z","iopub.status.idle":"2025-06-12T16:10:28.171824Z","shell.execute_reply.started":"2025-06-12T16:10:26.660802Z","shell.execute_reply":"2025-06-12T16:10:28.170889Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_correlation_top = pd.concat([target_correlation.head(5), target_correlation.tail(5)])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T16:10:28.172845Z","iopub.execute_input":"2025-06-12T16:10:28.173099Z","iopub.status.idle":"2025-06-12T16:10:28.178670Z","shell.execute_reply.started":"2025-06-12T16:10:28.173081Z","shell.execute_reply":"2025-06-12T16:10:28.177759Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_relationship(list(zip(target_correlation_top.index, ['label']*len(target_correlation_top), target_correlation_top['Corr'])), df_train, cols=5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T16:10:28.179680Z","iopub.execute_input":"2025-06-12T16:10:28.180765Z","iopub.status.idle":"2025-06-12T16:10:32.433333Z","shell.execute_reply.started":"2025-06-12T16:10:28.180637Z","shell.execute_reply":"2025-06-12T16:10:32.432273Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Target Deep Dive","metadata":{}},{"cell_type":"code","source":"fig = plt.figure(figsize=(14,6))\n\ngs = GridSpec(3, 2, figure=fig)\nax1 = fig.add_subplot(gs[0, :])\nax2 = fig.add_subplot(gs[1, :])\nax3 = fig.add_subplot(gs[2, 0])\nax4 = fig.add_subplot(gs[2, 1])\n\nax1.plot(df_train['label'])\nax2.plot(df_train['label'].cumsum())\nax3.hist(df_train['label'], bins=81)\nax4.boxplot(df_train['label'], vert=False, patch_artist=True)\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T16:10:32.434717Z","iopub.execute_input":"2025-06-12T16:10:32.435038Z","iopub.status.idle":"2025-06-12T16:10:33.727033Z","shell.execute_reply.started":"2025-06-12T16:10:32.434986Z","shell.execute_reply":"2025-06-12T16:10:33.725925Z"}},"outputs":[],"execution_count":null}]}