{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84493,"databundleVersionId":9849268,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# plotting stuff\nfrom pandas.plotting import lag_plot\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nimport plotly.graph_objects as go\ncolorMap = sns.light_palette(\"blue\", as_cmap=True)\n\n\n# system\nimport warnings\nwarnings.filterwarnings('ignore')\n# for the image import\n\nfrom IPython.display import Image\n# garbage collector to keep RAM in check\nimport gc \n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-16T05:23:12.964392Z","iopub.execute_input":"2024-10-16T05:23:12.964685Z","iopub.status.idle":"2024-10-16T05:23:15.861595Z","shell.execute_reply.started":"2024-10-16T05:23:12.964652Z","shell.execute_reply":"2024-10-16T05:23:15.860690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"author = {Keshav Kumar},\n\ntitle = {Jane Street Real-Time Market Data Forecasting},\n\npublisher = {Kaggle},\n\nyear = {2024},\nurl = {https://www.kaggle.com/competitions/jane-street-real-time-market-data-forecasting}","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:24:18.643062Z","iopub.execute_input":"2024-10-16T05:24:18.644015Z","iopub.status.idle":"2024-10-16T05:24:18.650633Z","shell.execute_reply.started":"2024-10-16T05:24:18.643969Z","shell.execute_reply":"2024-10-16T05:24:18.649420Z"}}},{"cell_type":"code","source":"resp = pd.read_csv('/kaggle/input/jane-street-real-time-market-data-forecasting/responders.csv')\nresp.tail()","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:24:41.822676Z","iopub.execute_input":"2024-10-16T05:24:41.823553Z","iopub.status.idle":"2024-10-16T05:24:41.851771Z","shell.execute_reply.started":"2024-10-16T05:24:41.823509Z","shell.execute_reply":"2024-10-16T05:24:41.850853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Features file, index_col=0 so the index is feature column\n\nfeat = pd.read_csv('/kaggle/input/jane-street-real-time-market-data-forecasting/features.csv', index_col=0)\nfeat.tail()","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:24:51.559222Z","iopub.execute_input":"2024-10-16T05:24:51.559580Z","iopub.status.idle":"2024-10-16T05:24:51.586578Z","shell.execute_reply.started":"2024-10-16T05:24:51.559540Z","shell.execute_reply":"2024-10-16T05:24:51.585673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#By Safadoost\n#encode the boolean  True/False to plot a heatmap after\n\ncategorical_columns = feat[:]\n# Create a new column for each unique value in the categorical columns.\ndf_categorical = pd.get_dummies(\n    data = feat ,\n    prefix = 'OHE' ,  # One hot encoding\n    prefix_sep = '_' ,\n    columns = categorical_columns ,\n    #drop_first = True ,\n    dtype = 'int8'\n)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:28:10.790943Z","iopub.execute_input":"2024-10-16T05:28:10.791929Z","iopub.status.idle":"2024-10-16T05:28:10.818424Z","shell.execute_reply.started":"2024-10-16T05:28:10.791875Z","shell.execute_reply":"2024-10-16T05:28:10.817463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,10))\nsns.heatmap(df_categorical.corr(),annot=True);","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:28:20.760439Z","iopub.execute_input":"2024-10-16T05:28:20.760832Z","iopub.status.idle":"2024-10-16T05:28:22.016768Z","shell.execute_reply.started":"2024-10-16T05:28:20.760777Z","shell.execute_reply":"2024-10-16T05:28:22.015743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=4/part-0.parquet\")\ntrain_data.tail()","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:28:46.970875Z","iopub.execute_input":"2024-10-16T05:28:46.971244Z","iopub.status.idle":"2024-10-16T05:28:59.043245Z","shell.execute_reply.started":"2024-10-16T05:28:46.971209Z","shell.execute_reply":"2024-10-16T05:28:59.042224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(15, 5))\nbalance= pd.Series(train_data['responder_0']).cumsum()\nax.set_xlabel (\"Trade\", fontsize=18)\nax.set_ylabel (\"Cumulative responder 0\", fontsize=18);\nbalance.plot(lw=3);\ndel balance\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:29:10.121792Z","iopub.execute_input":"2024-10-16T05:29:10.122197Z","iopub.status.idle":"2024-10-16T05:29:11.369107Z","shell.execute_reply.started":"2024-10-16T05:29:10.122159Z","shell.execute_reply":"2024-10-16T05:29:11.368154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(15, 5))\nbalance= pd.Series(train_data['responder_0']).cumsum()\nresp_1= pd.Series(train_data['responder_1']).cumsum()\nresp_2= pd.Series(train_data['responder_2']).cumsum()\nresp_3= pd.Series(train_data['responder_3']).cumsum()\nresp_4= pd.Series(train_data['responder_4']).cumsum()\nax.set_xlabel (\"Trade\", fontsize=18)\nax.set_title (\"Cumulative resp and time horizons 1, 2, 3, and 4 (500 days)\", fontsize=18)\nbalance.plot(lw=3)\nresp_1.plot(lw=3)\nresp_2.plot(lw=3)\nresp_3.plot(lw=3)\nresp_4.plot(lw=3)\nplt.legend(loc=\"upper left\");\ndel resp_1\ndel resp_2\ndel resp_3\ndel resp_4\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:29:36.715998Z","iopub.execute_input":"2024-10-16T05:29:36.716960Z","iopub.status.idle":"2024-10-16T05:29:41.123243Z","shell.execute_reply.started":"2024-10-16T05:29:36.716914Z","shell.execute_reply":"2024-10-16T05:29:41.122256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (12,5))\nax = sns.distplot(train_data['responder_0'], \n             bins=30, #Original was 3000 bins\n             kde_kws={\"clip\":(-0.05,0.05)}, \n             hist_kws={\"range\":(-0.05,0.05)},\n             color='darkcyan', \n             kde=False);\nvalues = np.array([rec.get_height() for rec in ax.patches])\nnorm = plt.Normalize(values.min(), values.max())\ncolors = plt.cm.jet(norm(values))\nfor rec, col in zip(ax.patches, colors):\n    rec.set_color(col)\nplt.xlabel(\"Histogram of the resp values\", size=14)\nplt.show();\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:30:01.671761Z","iopub.execute_input":"2024-10-16T05:30:01.672591Z","iopub.status.idle":"2024-10-16T05:30:02.110805Z","shell.execute_reply.started":"2024-10-16T05:30:01.672551Z","shell.execute_reply":"2024-10-16T05:30:02.109882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"min_resp = train_data['responder_0'].min()\nprint('The minimum value for resp is: %.5f' % min_resp)\nmax_resp = train_data['responder_0'].max()\nprint('The maximum value for resp is:  %.5f' % max_resp)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:30:14.025675Z","iopub.execute_input":"2024-10-16T05:30:14.026926Z","iopub.status.idle":"2024-10-16T05:30:14.046414Z","shell.execute_reply.started":"2024-10-16T05:30:14.026872Z","shell.execute_reply":"2024-10-16T05:30:14.045477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"min_resp = train_data['responder_0'].min()\nprint('The minimum value for resp is: %.5f' % min_resp)\nmax_resp = train_data['responder_0'].max()\nprint('The maximum value for resp is:  %.5f' % max_resp)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:30:22.866245Z","iopub.execute_input":"2024-10-16T05:30:22.866608Z","iopub.status.idle":"2024-10-16T05:30:22.887458Z","shell.execute_reply.started":"2024-10-16T05:30:22.866571Z","shell.execute_reply":"2024-10-16T05:30:22.885619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"values = np.array([rec.get_height() for rec in ax.patches])\n\nfrom scipy.optimize import curve_fit\n# the values\nx = list(range(len(values)))\nx = [((i)-15)/30 for i in x]\ny = values\n\ndef Lorentzian(x, x0, gamma, A):\n    return A * gamma**2/(gamma**2+( x - x0 )**2)\n\n# seed guess\ninitial_guess=(0, 0.001, 30)\n\nparameters,covariance=curve_fit(Lorentzian,x,y,initial_guess)\nsigma=np.sqrt(np.diag(covariance))\n\n# and plot\nplt.figure(figsize = (12,5))\nax = sns.distplot(train_data['responder_0'], \n             bins=30, \n             kde_kws={\"clip\":(-0.05,0.05)}, \n             hist_kws={\"range\":(-0.05,0.05)},\n             color='darkcyan', \n             kde=False);\nvalues = np.array([rec.get_height() for rec in ax.patches])\nplt.xlabel(\"Histogram of the responder_0 values\", size=14)\nplt.plot(x,Lorentzian(x,*parameters),'--',color='black',lw=3)\nplt.show();\ndel values\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:31:23.827365Z","iopub.execute_input":"2024-10-16T05:31:23.827725Z","iopub.status.idle":"2024-10-16T05:31:24.254921Z","shell.execute_reply.started":"2024-10-16T05:31:23.827689Z","shell.execute_reply":"2024-10-16T05:31:24.254007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"percent_zeros = (100/train_data.shape[0])*((train_data.weight.values == 0).sum())\nprint('Percentage of zero weights is: %i' % percent_zeros +\"%\")","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:31:42.499519Z","iopub.execute_input":"2024-10-16T05:31:42.499913Z","iopub.status.idle":"2024-10-16T05:31:42.510079Z","shell.execute_reply.started":"2024-10-16T05:31:42.499874Z","shell.execute_reply":"2024-10-16T05:31:42.509055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"min_weight = train_data['weight'].min()\nprint('The minimum weight is: %.2f' % min_weight)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:31:52.334692Z","iopub.execute_input":"2024-10-16T05:31:52.335094Z","iopub.status.idle":"2024-10-16T05:31:52.347159Z","shell.execute_reply.started":"2024-10-16T05:31:52.335055Z","shell.execute_reply":"2024-10-16T05:31:52.345992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_weight = train_data['weight'].max()\nprint('The maximum weight was: %.2f' % max_weight)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:32:02.288734Z","iopub.execute_input":"2024-10-16T05:32:02.289148Z","iopub.status.idle":"2024-10-16T05:32:02.302059Z","shell.execute_reply.started":"2024-10-16T05:32:02.289108Z","shell.execute_reply":"2024-10-16T05:32:02.301111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data[train_data['weight']==train_data['weight'].max()]","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:32:09.566184Z","iopub.execute_input":"2024-10-16T05:32:09.566554Z","iopub.status.idle":"2024-10-16T05:32:09.607473Z","shell.execute_reply.started":"2024-10-16T05:32:09.566516Z","shell.execute_reply":"2024-10-16T05:32:09.606569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (12,5))\nax = sns.distplot(train_data['weight'], \n             bins=140, #Original is 1400\n             kde_kws={\"clip\":(0.001,1.4)}, \n             hist_kws={\"range\":(0.001,1.4)},\n             color='darkcyan', \n             kde=False);\nvalues = np.array([rec.get_height() for rec in ax.patches])\nnorm = plt.Normalize(values.min(), values.max())\ncolors = plt.cm.jet(norm(values))\nfor rec, col in zip(ax.patches, colors):\n    rec.set_color(col)\nplt.xlabel(\"Histogram of non-zero weights\", size=14)\nplt.show();\ndel values\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:32:21.553746Z","iopub.execute_input":"2024-10-16T05:32:21.554519Z","iopub.status.idle":"2024-10-16T05:32:22.189640Z","shell.execute_reply.started":"2024-10-16T05:32:21.554474Z","shell.execute_reply":"2024-10-16T05:32:22.188714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_nonZero = train_data.query('weight > 0').reset_index(drop = True)\nplt.figure(figsize = (10,4))\nax = sns.distplot(np.log(train_data_nonZero['weight']), \n             bins=100, #Original was 1000\n             kde_kws={\"clip\":(-4,5)}, \n             hist_kws={\"range\":(-4,5)},\n             color='darkcyan', \n             kde=False);\nvalues = np.array([rec.get_height() for rec in ax.patches])\nnorm = plt.Normalize(values.min(), values.max())\ncolors = plt.cm.jet(norm(values))\nfor rec, col in zip(ax.patches, colors):\n    rec.set_color(col)\nplt.xlabel(\"Histogram of the logarithm of the non-zero weights\", size=14)\nplt.show();\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:32:35.391819Z","iopub.execute_input":"2024-10-16T05:32:35.392219Z","iopub.status.idle":"2024-10-16T05:32:37.536458Z","shell.execute_reply.started":"2024-10-16T05:32:35.392181Z","shell.execute_reply":"2024-10-16T05:32:37.535481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"values = np.array([rec.get_height() for rec in ax.patches])\n\nfrom scipy.optimize import curve_fit\n# the values\nx = list(range(len(values)))\nx = [(i/11)-4 for i in x]  #Original was 110\ny = values\n\n# define a Gaussian function\ndef Gaussian(x,mu,sigma,A):\n    return A*np.exp(-0.5 * ((x-mu)/sigma)**2)\n\ndef bimodal(x,mu_1,sigma_1,A_1,mu_2,sigma_2,A_2):\n    return Gaussian(x,mu_1,sigma_1,A_1) + Gaussian(x,mu_2,sigma_2,A_2)\n# seed guess\ninitial_guess=(1, 1 , 1,    1, 1, 1)\n\n# the fit\nparameters,covariance=curve_fit(bimodal,x,y,initial_guess)\nsigma=np.sqrt(np.diag(covariance))\n\n# the plot\nplt.figure(figsize = (10,4))\nax = sns.distplot(np.log(train_data_nonZero['weight']), \n             bins=100,     #Original was 1000\n             kde_kws={\"clip\":(-4,5)}, \n             hist_kws={\"range\":(-4,5)},\n             color='darkcyan', \n             kde=False);\nvalues = np.array([rec.get_height() for rec in ax.patches])\nnorm = plt.Normalize(values.min(), values.max())\ncolors = plt.cm.jet(norm(values))\nfor rec, col in zip(ax.patches, colors):\n    rec.set_color(col)\nplt.xlabel(\"Histogram of the logarithm of the non-zero weights\", size=14)\n# plot gaussian #1\nplt.plot(x,Gaussian(x,parameters[0],parameters[1],parameters[2]),':',color='black',lw=2,label='Gaussian #1', alpha=0.8)\n# plot gaussian #2\nplt.plot(x,Gaussian(x,parameters[3],parameters[4],parameters[5]),'--',color='black',lw=2,label='Gaussian #2', alpha=0.8)\n# plot the two gaussians together\nplt.plot(x,bimodal(x,*parameters),color='black',lw=2, alpha=0.7)\nplt.legend(loc=\"upper left\");\nplt.show();\ndel values \ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:33:38.753735Z","iopub.execute_input":"2024-10-16T05:33:38.754346Z","iopub.status.idle":"2024-10-16T05:33:39.394689Z","shell.execute_reply.started":"2024-10-16T05:33:38.754299Z","shell.execute_reply":"2024-10-16T05:33:39.393860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['feature_00']   = train_data['weight']*train_data['responder_0']\ntrain_data['feature_01'] = train_data['weight']*train_data['responder_1']\ntrain_data['feature_02'] = train_data['weight']*train_data['responder_2']\ntrain_data['feature_03'] = train_data['weight']*train_data['responder_3']\ntrain_data['feature_04'] = train_data['weight']*train_data['responder_4']\nfig, ax = plt.subplots(figsize=(15, 5))\nresp    = pd.Series(1+(train_data.groupby('date_id')['feature_00'].mean())).cumprod()\nresp_1  = pd.Series(1+(train_data.groupby('date_id')['feature_01'].mean())).cumprod()\nresp_2  = pd.Series(1+(train_data.groupby('date_id')['feature_02'].mean())).cumprod()\nresp_3  = pd.Series(1+(train_data.groupby('date_id')['feature_03'].mean())).cumprod()\nresp_4  = pd.Series(1+(train_data.groupby('date_id')['feature_04'].mean())).cumprod()\nax.set_xlabel (\"Day\", fontsize=18)\nax.set_title (\"Cumulative daily return for responder and time horizons 1, 2, 3, and 4 (500 days)\", fontsize=18)\nresp.plot(lw=3, label='responder_0 x weight')\nresp_1.plot(lw=3, label='responder_1 x weight')\nresp_2.plot(lw=3, label='responder_2 x weight')\nresp_3.plot(lw=3, label='responder_3 x weight')\nresp_4.plot(lw=3, label='responder_4 x weight')\n# day 85 marker\nax.axvline(x=85, linestyle='--', alpha=0.3, c='red', lw=1)\nax.axvspan(0, 85 , color=sns.xkcd_rgb['grey'], alpha=0.1)\nplt.legend(loc=\"lower left\");","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:34:07.025263Z","iopub.execute_input":"2024-10-16T05:34:07.025646Z","iopub.status.idle":"2024-10-16T05:34:07.924737Z","shell.execute_reply.started":"2024-10-16T05:34:07.025607Z","shell.execute_reply":"2024-10-16T05:34:07.923822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_no_0 = train_data.query('weight > 0').reset_index(drop = True)\ntrain_data_no_0['wAbsResp'] = train_data_no_0['weight'] * (train_data_no_0['responder_0'])\n#plot\nplt.figure(figsize = (12,5))\nax = sns.distplot(train_data_no_0['wAbsResp'], \n             bins=150,  #Original was 1500\n             kde_kws={\"clip\":(-0.02,0.02)}, \n             hist_kws={\"range\":(-0.02,0.02)},\n             color='darkcyan', \n             kde=False);\nvalues = np.array([rec.get_height() for rec in ax.patches])\nnorm = plt.Normalize(values.min(), values.max())\ncolors = plt.cm.jet(norm(values))\nfor rec, col in zip(ax.patches, colors):\n    rec.set_color(col)\nplt.xlabel(\"Histogram of the weights * resp\", size=14)\nplt.show();","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:34:30.158420Z","iopub.execute_input":"2024-10-16T05:34:30.158813Z","iopub.status.idle":"2024-10-16T05:34:33.596513Z","shell.execute_reply.started":"2024-10-16T05:34:30.158776Z","shell.execute_reply":"2024-10-16T05:34:33.595494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['feature_00'].value_counts()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:34:42.523659Z","iopub.execute_input":"2024-10-16T05:34:42.524047Z","iopub.status.idle":"2024-10-16T05:34:43.749820Z","shell.execute_reply.started":"2024-10-16T05:34:42.524008Z","shell.execute_reply":"2024-10-16T05:34:43.748856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(15, 4))\nfeature_0 = pd.Series(train_data['feature_00']).cumsum()\nax.set_xlabel (\"Trade\", fontsize=18)\nax.set_ylabel (\"feature_00 (cumulative)\", fontsize=18);\nfeature_0.plot(lw=3);","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:34:54.275133Z","iopub.execute_input":"2024-10-16T05:34:54.275950Z","iopub.status.idle":"2024-10-16T05:34:55.362964Z","shell.execute_reply.started":"2024-10-16T05:34:54.275909Z","shell.execute_reply":"2024-10-16T05:34:55.362022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfig, ((ax1, ax2), (ax3, ax4)) = plt.subplots(2, 2,figsize=(20,10))\n\nax1.plot((pd.Series(train_data['feature_01']).cumsum()), lw=3, color='red')\nax1.set_title (\"Noisy\", fontsize=22);\nax1.axvline(x=514052, linestyle='--', alpha=0.3, c='green', lw=2)\nax1.axvspan(0, 514052 , color=sns.xkcd_rgb['grey'], alpha=0.1)\nax1.set_xlim(xmin=0)\nax1.set_ylabel (\"feature_01\", fontsize=18);\n\nax2.plot((pd.Series(train_data['feature_03']).cumsum()), lw=3, color='green')\nax2.set_title (\"Negative\", fontsize=22);\nax2.axvline(x=514052, linestyle='--', alpha=0.3, c='red', lw=2)\nax2.axvspan(0, 514052 , color=sns.xkcd_rgb['grey'], alpha=0.1)\nax2.set_xlim(xmin=0)\nax2.set_ylabel (\"feature_03\", fontsize=18);\n\nax3.plot((pd.Series(train_data['feature_55']).cumsum()), lw=3, color='darkorange')\nax3.set_title (\"Hybryd (Tag 21)\", fontsize=22);\nax3.set_xlabel (\"Trade\", fontsize=18)\nax3.axvline(x=514052, linestyle='--', alpha=0.3, c='green', lw=2)\nax3.axvspan(0, 514052 , color=sns.xkcd_rgb['grey'], alpha=0.1)\nax3.set_xlim(xmin=0)\nax3.set_ylabel (\"feature_55\", fontsize=18);\n\nax4.plot((pd.Series(train_data['feature_73']).cumsum()), lw=3, color='blue')\nax4.set_title (\"Hybryd\", fontsize=22)\nax4.set_xlabel (\"Trade\", fontsize=18)\nax4.set_ylabel (\"feature_73\", fontsize=18);\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:35:22.996719Z","iopub.execute_input":"2024-10-16T05:35:22.997102Z","iopub.status.idle":"2024-10-16T05:35:27.012080Z","shell.execute_reply.started":"2024-10-16T05:35:22.997065Z","shell.execute_reply":"2024-10-16T05:35:27.011196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfig, ax = plt.subplots(figsize=(15, 5))\nfeature_60= pd.Series(train_data['feature_60']).cumsum()\nfeature_61= pd.Series(train_data['feature_61']).cumsum()\nfeature_62= pd.Series(train_data['feature_62']).cumsum()\nfeature_63= pd.Series(train_data['feature_63']).cumsum()\nfeature_64= pd.Series(train_data['feature_64']).cumsum()\nfeature_65= pd.Series(train_data['feature_65']).cumsum()\nfeature_66= pd.Series(train_data['feature_66']).cumsum()\nfeature_67= pd.Series(train_data['feature_67']).cumsum()\nfeature_68= pd.Series(train_data['feature_68']).cumsum()\n#feature_69= pd.Series(train_data['feature_69']).cumsum()\nax.set_xlabel (\"Trade\", fontsize=18)\nax.set_title (\"Cumulative plot for feature_60 ... feature_68.\", fontsize=18)\nfeature_60.plot(lw=3)\nfeature_61.plot(lw=3)\nfeature_62.plot(lw=3)\nfeature_63.plot(lw=3)\nfeature_64.plot(lw=3)\nfeature_65.plot(lw=3)\nfeature_66.plot(lw=3)\nfeature_67.plot(lw=3)\nfeature_68.plot(lw=3)\n#feature_69.plot(lw=3)\nplt.legend(loc=\"upper left\");\ndel feature_60, feature_61, feature_62, feature_63, feature_64, feature_65, feature_66 ,feature_67, feature_68\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:35:45.754599Z","iopub.execute_input":"2024-10-16T05:35:45.755016Z","iopub.status.idle":"2024-10-16T05:35:53.041006Z","shell.execute_reply.started":"2024-10-16T05:35:45.754974Z","shell.execute_reply":"2024-10-16T05:35:53.040088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nsns.set_palette(\"bright\")\n\nfig, axes = plt.subplots(2,2,figsize=(8,8))\n\nsns.distplot(train_data[['feature_60']], hist=True, bins=200,  ax=axes[0,0])\nsns.distplot(train_data[['feature_61']], hist=True, bins=200,  ax=axes[0,0])\naxes[0,0].set_title (\"features 60 and 61\", fontsize=18)\naxes[0,0].legend(labels=['60', '61'])\n\nsns.distplot(train_data[['feature_62']], hist=True,  bins=200, ax=axes[0,1])\nsns.distplot(train_data[['feature_63']], hist=True,  bins=200, ax=axes[0,1])\naxes[0,1].set_title (\"features 62 and 63\", fontsize=18)\naxes[0,1].legend(labels=['62', '63'])\n\nsns.distplot(train_data[['feature_65']], hist=True,  bins=200, ax=axes[1,0])\nsns.distplot(train_data[['feature_66']], hist=True,  bins=200, ax=axes[1,0])\naxes[1,0].set_title (\"features 65 and 66\", fontsize=18)\naxes[1,0].legend(labels=['65', '66'])\n\n\nsns.distplot(train_data[['feature_67']], hist=True,  bins=200, ax=axes[1,1])\nsns.distplot(train_data[['feature_68']], hist=True,  bins=200, ax=axes[1,1])\naxes[1,1].set_title (\"features 67 and 68\", fontsize=18)\naxes[1,1].legend(labels=['67', '68'])\n\nplt.show();\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:36:06.787464Z","iopub.execute_input":"2024-10-16T05:36:06.788359Z","iopub.status.idle":"2024-10-16T05:38:38.165929Z","shell.execute_reply.started":"2024-10-16T05:36:06.788317Z","shell.execute_reply":"2024-10-16T05:38:38.164851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nplt.figure(figsize = (12,5))\nax = sns.distplot(train_data['feature_64'], \n             bins=120, \n             kde_kws={\"clip\":(-6,6)}, \n             hist_kws={\"range\":(-6,6)},\n             color='darkcyan', \n             kde=False);\nvalues = np.array([rec.get_height() for rec in ax.patches])\nnorm = plt.Normalize(values.min(), values.max())\ncolors = plt.cm.jet(norm(values))\nfor rec, col in zip(ax.patches, colors):\n    rec.set_color(col)\nplt.xlabel(\"Histogram of feature_64\", size=14)\nplt.show();\ndel values\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:48:15.884325Z","iopub.execute_input":"2024-10-16T05:48:15.884709Z","iopub.status.idle":"2024-10-16T05:48:16.482263Z","shell.execute_reply.started":"2024-10-16T05:48:15.884671Z","shell.execute_reply":"2024-10-16T05:48:16.481350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(15, 4))\nax.scatter(train_data_nonZero.weight, train_data_nonZero.feature_51, s=0.1, color='b')\nax.set_xlabel('weight')\nax.set_ylabel('feature_51')\nplt.show();","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:48:19.078518Z","iopub.execute_input":"2024-10-16T05:48:19.079240Z","iopub.status.idle":"2024-10-16T05:48:21.072643Z","shell.execute_reply.started":"2024-10-16T05:48:19.079201Z","shell.execute_reply":"2024-10-16T05:48:21.071775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfig, ax = plt.subplots(figsize=(15, 5))\nfeature_55= pd.Series(train_data['feature_55']).cumsum()\nfeature_56= pd.Series(train_data['feature_56']).cumsum()\nfeature_57= pd.Series(train_data['feature_57']).cumsum()\nfeature_58= pd.Series(train_data['feature_58']).cumsum()\nfeature_59= pd.Series(train_data['feature_59']).cumsum()\nax.set_xlabel (\"Trade\", fontsize=18)\nax.set_title (\"Cumulative plot for the 'Tag 21' features (55-59)\", fontsize=18)\nax.axvline(x=514052, linestyle='--', alpha=0.3, c='black', lw=1)\nax.axvspan(0,  514052, color=sns.xkcd_rgb['grey'], alpha=0.1)\nfeature_55.plot(lw=3)\nfeature_56.plot(lw=3)\nfeature_57.plot(lw=3)\nfeature_58.plot(lw=3)\nfeature_59.plot(lw=3)\nplt.legend(loc=\"upper left\");\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-10-16T05:48:23.791231Z","iopub.execute_input":"2024-10-16T05:48:23.791611Z","iopub.status.idle":"2024-10-16T05:48:28.217059Z","shell.execute_reply.started":"2024-10-16T05:48:23.791572Z","shell.execute_reply":"2024-10-16T05:48:28.216112Z"},"trusted":true},"execution_count":null,"outputs":[]}]}