{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 🧾Table of Contents\n* [Main Features](#features)\n* [Anonymized Features](#features_x)\n* [Target](#target)\n* [Target split in upside/downside](#split)","metadata":{}},{"cell_type":"code","source":"# install package for distribution fitting\n!pip install fitter","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:44:59.810488Z","iopub.execute_input":"2025-06-07T16:44:59.810823Z","iopub.status.idle":"2025-06-07T16:45:06.400796Z","shell.execute_reply.started":"2025-06-07T16:44:59.810797Z","shell.execute_reply":"2025-06-07T16:45:06.399343Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# packages\n\n# standard\nimport numpy as np\nimport pandas as pd\nimport time\nimport gc\n\n# plots\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# statistics\nfrom fitter import Fitter, get_common_distributions, get_distributions\n\n# warnings\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:45:14.840051Z","iopub.execute_input":"2025-06-07T16:45:14.840398Z","iopub.status.idle":"2025-06-07T16:45:15.044477Z","shell.execute_reply.started":"2025-06-07T16:45:14.840366Z","shell.execute_reply":"2025-06-07T16:45:15.043506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# configs \n\n# show all columns\npd.set_option('display.max_columns', 500)\n\n# colors\ndefault_color_1 = 'darkblue'\ndefault_color_2 = 'darkgreen'\ndefault_color_3 = 'darkred'\n\n# random seed\nrandom_seed = 123","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:45:17.688548Z","iopub.execute_input":"2025-06-07T16:45:17.689275Z","iopub.status.idle":"2025-06-07T16:45:17.694976Z","shell.execute_reply.started":"2025-06-07T16:45:17.689243Z","shell.execute_reply":"2025-06-07T16:45:17.693754Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# load training data\nt1 = time.time()\ndf_train = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\nt2 = time.time()\nprint('Elapsed time [s]:', np.round(t2-t1, 2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:45:20.387624Z","iopub.execute_input":"2025-06-07T16:45:20.388007Z","iopub.status.idle":"2025-06-07T16:45:47.314974Z","shell.execute_reply.started":"2025-06-07T16:45:20.387981Z","shell.execute_reply":"2025-06-07T16:45:47.313875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# preview\ndf_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:45:50.656159Z","iopub.execute_input":"2025-06-07T16:45:50.656506Z","iopub.status.idle":"2025-06-07T16:45:50.940845Z","shell.execute_reply.started":"2025-06-07T16:45:50.656480Z","shell.execute_reply":"2025-06-07T16:45:50.939559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# structure details\ndf_train.info(verbose=True, show_counts=True)","metadata":{"trusted":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2025-06-07T16:30:12.389021Z","iopub.execute_input":"2025-06-07T16:30:12.389263Z","iopub.status.idle":"2025-06-07T16:30:14.364590Z","shell.execute_reply.started":"2025-06-07T16:30:12.389242Z","shell.execute_reply":"2025-06-07T16:30:14.363564Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# main features\nfeatures_main = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']\n\n# anonymized features\nfeatures_x = ['X' + str(i) for i in range(1,890+1)]\n\n# target\ntarget = 'label'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:46:50.724375Z","iopub.execute_input":"2025-06-07T16:46:50.724719Z","iopub.status.idle":"2025-06-07T16:46:50.730562Z","shell.execute_reply.started":"2025-06-07T16:46:50.724690Z","shell.execute_reply":"2025-06-07T16:46:50.729268Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='features'></a>\n# Main Features","metadata":{}},{"cell_type":"code","source":"# basic stats main features\ndf_train[features_main].describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:30:14.371814Z","iopub.execute_input":"2025-06-07T16:30:14.372185Z","iopub.status.idle":"2025-06-07T16:30:14.547209Z","shell.execute_reply.started":"2025-06-07T16:30:14.372155Z","shell.execute_reply":"2025-06-07T16:30:14.546356Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot main features\nfor f in features_main:\n    fig, (ax1, ax2) = plt.subplots(2, 1, figsize=(10,6), sharex=True)\n    ax1.hist(df_train[f], bins=100, color=default_color_1)\n    ax1.grid()\n    ax1.set_title(f)\n    # for boxplot we need to remove the NaNs first\n    feature_wo_nan = df_train[~np.isnan(df_train[f])][f]\n    ax2.boxplot(feature_wo_nan, vert=False)\n    ax2.grid()\n    ax2.set_title(f + ' - boxplot')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:30:14.548115Z","iopub.execute_input":"2025-06-07T16:30:14.548358Z","iopub.status.idle":"2025-06-07T16:30:25.129927Z","shell.execute_reply.started":"2025-06-07T16:30:14.548338Z","shell.execute_reply":"2025-06-07T16:30:25.128496Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot main features as time series\nfor f in features_main:\n    plt.figure(figsize=(14,4))\n    plt.scatter(df_train.index, df_train[f], \n            color=default_color_1, s=1)\n    plt.title(f + ' (Time Series)')\n    plt.grid()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:30:25.134047Z","iopub.execute_input":"2025-06-07T16:30:25.134333Z","iopub.status.idle":"2025-06-07T16:30:28.008679Z","shell.execute_reply.started":"2025-06-07T16:30:25.134311Z","shell.execute_reply":"2025-06-07T16:30:28.007693Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# correlation\ncorr_pearson = df_train[features_main+[target]].corr(method='pearson')\ncorr_spearman = df_train[features_main+[target]].corr(method='spearman')\n\nplt.figure(figsize=(12,4))\nax1 = plt.subplot(1,2,1)\nsns.heatmap(corr_pearson, annot=True, cmap='RdYlGn', \n            vmin=-1, vmax=+1, fmt='.3f',\n            linecolor='black', linewidths=0.5)\nplt.title('Pearson Correlation')\n\nax2 = plt.subplot(1,2,2, sharex=ax1)\nsns.heatmap(corr_spearman, annot=True, cmap='RdYlGn', \n            vmin=-1, vmax=+1, fmt='.3f',\n            linecolor='black', linewidths=0.5)\nplt.title('Spearman Correlation')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:30:28.009623Z","iopub.execute_input":"2025-06-07T16:30:28.010176Z","iopub.status.idle":"2025-06-07T16:30:29.594722Z","shell.execute_reply.started":"2025-06-07T16:30:28.010153Z","shell.execute_reply":"2025-06-07T16:30:29.593644Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# garbage collection\ngc.collect();","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:30:29.595788Z","iopub.execute_input":"2025-06-07T16:30:29.596131Z","iopub.status.idle":"2025-06-07T16:30:29.709094Z","shell.execute_reply.started":"2025-06-07T16:30:29.596101Z","shell.execute_reply":"2025-06-07T16:30:29.707597Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='features_x'></a>\n# Anonymized Features","metadata":{}},{"cell_type":"code","source":"# calc basic stats for anonymized features\ndf_train[features_x].describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:30:29.710334Z","iopub.execute_input":"2025-06-07T16:30:29.710699Z","iopub.status.idle":"2025-06-07T16:30:50.433399Z","shell.execute_reply.started":"2025-06-07T16:30:29.710651Z","shell.execute_reply":"2025-06-07T16:30:50.432346Z"},"_kg_hide-output":false},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# garbage collection\ngc.collect();","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:30:50.434452Z","iopub.execute_input":"2025-06-07T16:30:50.434733Z","iopub.status.idle":"2025-06-07T16:30:50.542003Z","shell.execute_reply.started":"2025-06-07T16:30:50.434711Z","shell.execute_reply":"2025-06-07T16:30:50.541011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# boxplot of all anonymized variables\nfor i in range(17):\n    print('Columns', 50*i+1 , 'to', 50*i+50)\n    df_train[features_x].iloc[:,(50*i):(50*i+50)].plot(kind='box', figsize=(15,5))\n    plt.xticks(rotation=90)\n    plt.grid()\n    plt.show()\n\n# separate plot for reminaing columns\nprint('Columns', 851 , 'to', 890)\ndf_train[features_x].iloc[:,850:(850+1+50)].plot(kind='box', figsize=(15,5))\nplt.xticks(rotation=90)\nplt.grid()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:30:50.542950Z","iopub.execute_input":"2025-06-07T16:30:50.543278Z","iopub.status.idle":"2025-06-07T16:33:13.187175Z","shell.execute_reply.started":"2025-06-07T16:30:50.543251Z","shell.execute_reply":"2025-06-07T16:33:13.186236Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# garbage collection\ngc.collect();","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:46:19.849684Z","iopub.execute_input":"2025-06-07T16:46:19.850012Z","iopub.status.idle":"2025-06-07T16:46:19.949522Z","shell.execute_reply.started":"2025-06-07T16:46:19.849988Z","shell.execute_reply":"2025-06-07T16:46:19.948357Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### 💡 For the columns X697..X717 no plot points are displayed. Reason is that for those columns there are only infinite values.","metadata":{}},{"cell_type":"code","source":"df_train['X697'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:33:13.421223Z","iopub.execute_input":"2025-06-07T16:33:13.421533Z","iopub.status.idle":"2025-06-07T16:33:13.450147Z","shell.execute_reply.started":"2025-06-07T16:33:13.421511Z","shell.execute_reply":"2025-06-07T16:33:13.449272Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### 💡 Furthermore there are columns with only 0s, namely X864, X867 and X869..X872","metadata":{}},{"cell_type":"code","source":"df_train['X864'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:33:13.451014Z","iopub.execute_input":"2025-06-07T16:33:13.451253Z","iopub.status.idle":"2025-06-07T16:33:13.476652Z","shell.execute_reply.started":"2025-06-07T16:33:13.451228Z","shell.execute_reply":"2025-06-07T16:33:13.475769Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### All those columns can be removed for model training.","metadata":{}},{"cell_type":"code","source":"# define columns to be removed\ndrop_cols = ['X' + str(i) for i in range(697,717+1)]\ndrop_cols = drop_cols + ['X864','X867','X869','X870','X871','X872']\nprint(drop_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:33:13.477513Z","iopub.execute_input":"2025-06-07T16:33:13.477783Z","iopub.status.idle":"2025-06-07T16:33:13.497013Z","shell.execute_reply.started":"2025-06-07T16:33:13.477763Z","shell.execute_reply":"2025-06-07T16:33:13.496066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# remove redundant columns\ndf_train.drop(drop_cols, axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:33:13.498047Z","iopub.execute_input":"2025-06-07T16:33:13.498419Z","iopub.status.idle":"2025-06-07T16:33:14.667812Z","shell.execute_reply.started":"2025-06-07T16:33:13.498393Z","shell.execute_reply":"2025-06-07T16:33:14.666912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# garbage collection\ngc.collect();","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:46:25.624876Z","iopub.execute_input":"2025-06-07T16:46:25.625188Z","iopub.status.idle":"2025-06-07T16:46:25.722494Z","shell.execute_reply.started":"2025-06-07T16:46:25.625160Z","shell.execute_reply":"2025-06-07T16:46:25.721289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# adjust list of features accordingly\nfeatures_x = [x for x in features_x if x not in drop_cols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:33:14.781078Z","iopub.execute_input":"2025-06-07T16:33:14.781422Z","iopub.status.idle":"2025-06-07T16:33:14.798789Z","shell.execute_reply.started":"2025-06-07T16:33:14.781393Z","shell.execute_reply":"2025-06-07T16:33:14.797749Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Correlations","metadata":{}},{"cell_type":"code","source":"# calc correlations (this takes quite some time...)\nt1 = time.time()\ncorr_pearson_x = df_train[features_x].corr(method='pearson')\nt2 = time.time()\nprint('Elapsed time [s]:', np.round(t2-t1, 2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:33:14.818326Z","iopub.execute_input":"2025-06-07T16:33:14.818592Z","execution_failed":"2025-06-07T16:44:03.055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# create data frame to store all results\nn_features = len(features_x)\nn_rows = int(n_features*(n_features - 1) / 2)\ncorr_stats = pd.DataFrame(data=np.zeros((n_rows,3)), columns=['x','y','corr'])\ncorr_stats.x = corr_stats.x.astype(str)\ncorr_stats.y = corr_stats.y.astype(str)\n\n# rearrange all correlations in tabular form\nrow = 0\nfor i in range(n_features):\n    var_i = features_x[i]\n    for j in range(n_features):\n        if i<j:\n            var_j = features_x[j]\n            corr_x = corr_pearson_x.iloc[i,j]\n            # store results\n            corr_stats.loc[row,'x'] = var_i\n            corr_stats.loc[row,'y'] = var_j\n            corr_stats.loc[row,'corr'] = corr_x\n            row = row + 1\n\n# sort by correlation (descending)\ncorr_stats = corr_stats.sort_values(by=['corr'], ascending=False)\ncorr_stats = corr_stats.reset_index(drop=True)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-06-07T16:44:03.055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# top 10 correlations\ncorr_stats.head(50)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-06-07T16:44:03.056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# bottom 10 correlations\ncorr_stats.tail(25)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-06-07T16:44:03.056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot correlations\nplt.figure(figsize=(8,5))\nplt.hist(corr_stats['corr'], 100, color = default_color_1)\nplt.title('Feature Correlations')\nplt.grid()\nplt.show()","metadata":{"trusted":true,"execution":{"execution_failed":"2025-06-07T16:44:03.056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# export results\ncorr_stats.to_csv('correlation_stats.csv')","metadata":{"trusted":true,"execution":{"execution_failed":"2025-06-07T16:44:03.056Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='target'></a>\n# Target","metadata":{}},{"cell_type":"code","source":"# histogram of target\nplt.figure(figsize=(12,4))\nplt.hist(df_train[target], bins=1000, color=default_color_3)\nplt.title('Target (Histogram)')\nplt.grid()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:46:56.588550Z","iopub.execute_input":"2025-06-07T16:46:56.588895Z","iopub.status.idle":"2025-06-07T16:46:58.038635Z","shell.execute_reply.started":"2025-06-07T16:46:56.588871Z","shell.execute_reply":"2025-06-07T16:46:58.037540Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# boxplot of target\nplt.figure(figsize=(12,1))\nplt.boxplot(df_train[target], vert=False)\nplt.title('Target (Boxplot)')\nplt.grid()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:47:00.595421Z","iopub.execute_input":"2025-06-07T16:47:00.595773Z","iopub.status.idle":"2025-06-07T16:47:00.795297Z","shell.execute_reply.started":"2025-06-07T16:47:00.595747Z","shell.execute_reply":"2025-06-07T16:47:00.794249Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# target as time series\nplt.figure(figsize=(14,6))\nplt.scatter(df_train.index, df_train[target], \n            color=default_color_3, s=1, alpha=1)\nplt.title('Target (Time Series)')\nplt.grid()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:47:03.264480Z","iopub.execute_input":"2025-06-07T16:47:03.264806Z","iopub.status.idle":"2025-06-07T16:47:04.017027Z","shell.execute_reply.started":"2025-06-07T16:47:03.264781Z","shell.execute_reply":"2025-06-07T16:47:04.015631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# try to fit a few distribution types to target\n# we use a subset to achieve a reasonable run time\ndata = df_train[target].sample(50000, random_state=random_seed)\ndist_fitter = Fitter(data,\n                     distributions=['lognorm', 't', 'cauchy',\n                                    'genhyperbolic', 'norminvgauss', 'tukeylambda', \n                                    'gennorm', 'dgamma', 'johnsonsu'], \n                     timeout=300)\ndist_fitter.fit()\nplt.figure(figsize=(12,5))\ndist_fitter.summary(9)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:47:53.437982Z","iopub.execute_input":"2025-06-07T16:47:53.438313Z","iopub.status.idle":"2025-06-07T16:49:26.598583Z","shell.execute_reply.started":"2025-06-07T16:47:53.438293Z","shell.execute_reply":"2025-06-07T16:49:26.597533Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# just plotting the fitted PDFs\nplt.figure(figsize=(12,5))\ndist_fitter.plot_pdf(Nbest=9, lw=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:50:13.390495Z","iopub.execute_input":"2025-06-07T16:50:13.391359Z","iopub.status.idle":"2025-06-07T16:50:13.731230Z","shell.execute_reply.started":"2025-06-07T16:50:13.391315Z","shell.execute_reply":"2025-06-07T16:50:13.730033Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# best fit\ndist_fitter.get_best()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:50:31.582708Z","iopub.execute_input":"2025-06-07T16:50:31.583027Z","iopub.status.idle":"2025-06-07T16:50:31.590674Z","shell.execute_reply.started":"2025-06-07T16:50:31.583006Z","shell.execute_reply":"2025-06-07T16:50:31.589139Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='split'></a>\n# Target split in upside/downside","metadata":{}},{"cell_type":"code","source":"# split target in positive and negative values\nupside = df_train[target][df_train[target]>0]\ndnside = -df_train[target][df_train[target]<0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:51:02.474305Z","iopub.execute_input":"2025-06-07T16:51:02.474778Z","iopub.status.idle":"2025-06-07T16:51:02.494362Z","shell.execute_reply.started":"2025-06-07T16:51:02.474748Z","shell.execute_reply":"2025-06-07T16:51:02.493226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# stats for positive targets\nupside.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:51:03.873587Z","iopub.execute_input":"2025-06-07T16:51:03.874011Z","iopub.status.idle":"2025-06-07T16:51:03.898063Z","shell.execute_reply.started":"2025-06-07T16:51:03.873983Z","shell.execute_reply":"2025-06-07T16:51:03.896877Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# stats for negative targets (with sign switched)\ndnside.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:51:05.879467Z","iopub.execute_input":"2025-06-07T16:51:05.879805Z","iopub.status.idle":"2025-06-07T16:51:05.899360Z","shell.execute_reply.started":"2025-06-07T16:51:05.879780Z","shell.execute_reply":"2025-06-07T16:51:05.898113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# histogram of target upside\nplt.figure(figsize=(12,4))\nplt.hist(np.log10(upside), bins=200, color=default_color_3)\nplt.title('log10(Target Upside) (Histogram)')\nplt.grid()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:51:13.520148Z","iopub.execute_input":"2025-06-07T16:51:13.520468Z","iopub.status.idle":"2025-06-07T16:51:14.117754Z","shell.execute_reply.started":"2025-06-07T16:51:13.520444Z","shell.execute_reply":"2025-06-07T16:51:14.116703Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# histogram of target downside\nplt.figure(figsize=(12,4))\nplt.hist(np.log10(dnside), bins=200, color=default_color_3)\nplt.title('-log10(Target Downside) (Histogram)')\nplt.grid()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T16:51:17.858212Z","iopub.execute_input":"2025-06-07T16:51:17.858517Z","iopub.status.idle":"2025-06-07T16:51:18.273223Z","shell.execute_reply.started":"2025-06-07T16:51:17.858493Z","shell.execute_reply":"2025-06-07T16:51:18.271975Z"}},"outputs":[],"execution_count":null}]}