{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9849268,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Published on October 14, 2024. By Marília Prata, mpwolke.","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# plotting stuff\nfrom pandas.plotting import lag_plot\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nimport plotly.graph_objects as go\ncolorMap = sns.light_palette(\"blue\", as_cmap=True)\n\n# system\nimport warnings\nwarnings.filterwarnings('ignore')\n# for the image import\n\nfrom IPython.display import Image\n# garbage collector to keep RAM in check\nimport gc \n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-10-14T20:47:32.474756Z","iopub.execute_input":"2024-10-14T20:47:32.475666Z","iopub.status.idle":"2024-10-14T20:47:35.892185Z","shell.execute_reply.started":"2024-10-14T20:47:32.475622Z","shell.execute_reply":"2024-10-14T20:47:35.891102Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## After 3h:9m Plotting on Jane Street, I couldn't resist to add this cartoon\n\n![](https://pbs.twimg.com/media/GL_tr1vXYAAEV2H.jpg:large)","metadata":{}},{"cell_type":"markdown","source":"### Competition Citation\n\n@misc{jane-street-real-time-market-data-forecasting,\n\n    author = {Maanit Desai, Yirun Zhang, Ryan Holbrook, Kait O'Neil, Maggie Demkin},\n    \n    title = {Jane Street Real-Time Market Data Forecasting},\n    \n    publisher = {Kaggle},\n    \n    year = {2024},\n    url = {https://kaggle.com/competitions/jane-street-","metadata":{}},{"cell_type":"code","source":"resp = pd.read_csv('/kaggle/input/jane-street-real-time-market-data-forecasting/responders.csv')\nresp.tail()","metadata":{"execution":{"iopub.status.busy":"2024-10-14T20:55:49.800016Z","iopub.execute_input":"2024-10-14T20:55:49.800905Z","iopub.status.idle":"2024-10-14T20:55:49.815997Z","shell.execute_reply.started":"2024-10-14T20:55:49.800864Z","shell.execute_reply":"2024-10-14T20:55:49.815018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Features file, index_col=0 so the index is feature column\n\nfeat = pd.read_csv('/kaggle/input/jane-street-real-time-market-data-forecasting/features.csv', index_col=0)\nfeat.tail()","metadata":{"execution":{"iopub.status.busy":"2024-10-14T22:56:07.655794Z","iopub.execute_input":"2024-10-14T22:56:07.657043Z","iopub.status.idle":"2024-10-14T22:56:07.688118Z","shell.execute_reply.started":"2024-10-14T22:56:07.656984Z","shell.execute_reply":"2024-10-14T22:56:07.686988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#By Safadoost\n#encode the boolean  True/False to plot a heatmap after\n\ncategorical_columns = feat[:]\n# Create a new column for each unique value in the categorical columns.\ndf_categorical = pd.get_dummies(\n    data = feat ,\n    prefix = 'OHE' ,  # One hot encoding\n    prefix_sep = '_' ,\n    columns = categorical_columns ,\n    #drop_first = True ,\n    dtype = 'int8'\n)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T23:00:33.358299Z","iopub.execute_input":"2024-10-14T23:00:33.359279Z","iopub.status.idle":"2024-10-14T23:00:33.389274Z","shell.execute_reply.started":"2024-10-14T23:00:33.359227Z","shell.execute_reply":"2024-10-14T23:00:33.388193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Heatmap of the Features csv file  (booleans)","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12,10))\nsns.heatmap(df_categorical.corr(),annot=True);","metadata":{"execution":{"iopub.status.busy":"2024-10-14T23:02:48.067545Z","iopub.execute_input":"2024-10-14T23:02:48.068332Z","iopub.status.idle":"2024-10-14T23:02:49.289778Z","shell.execute_reply.started":"2024-10-14T23:02:48.068273Z","shell.execute_reply":"2024-10-14T23:02:49.288508Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Read One parquet file. Obviously, it's big.\n\ntrain_data = pd.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=4/part-0.parquet\")\ntrain_data.tail()","metadata":{"execution":{"iopub.status.busy":"2024-10-14T21:07:13.028757Z","iopub.execute_input":"2024-10-14T21:07:13.029450Z","iopub.status.idle":"2024-10-14T21:07:16.019067Z","shell.execute_reply.started":"2024-10-14T21:07:13.029407Z","shell.execute_reply":"2024-10-14T21:07:16.018095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's take a look at the cumulative values of responder_0 over time","metadata":{}},{"cell_type":"code","source":"#By Carl McBride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\nfig, ax = plt.subplots(figsize=(15, 5))\nbalance= pd.Series(train_data['responder_0']).cumsum()\nax.set_xlabel (\"Trade\", fontsize=18)\nax.set_ylabel (\"Cumulative responder 0\", fontsize=18);\nbalance.plot(lw=3);\ndel balance\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-10-14T21:08:56.155780Z","iopub.execute_input":"2024-10-14T21:08:56.157659Z","iopub.status.idle":"2024-10-14T21:08:57.639194Z","shell.execute_reply.started":"2024-10-14T21:08:56.157593Z","shell.execute_reply":"2024-10-14T21:08:57.638091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#By Carl McBride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\nfig, ax = plt.subplots(figsize=(15, 5))\nbalance= pd.Series(train_data['responder_0']).cumsum()\nresp_1= pd.Series(train_data['responder_1']).cumsum()\nresp_2= pd.Series(train_data['responder_2']).cumsum()\nresp_3= pd.Series(train_data['responder_3']).cumsum()\nresp_4= pd.Series(train_data['responder_4']).cumsum()\nax.set_xlabel (\"Trade\", fontsize=18)\nax.set_title (\"Cumulative resp and time horizons 1, 2, 3, and 4 (500 days)\", fontsize=18)\nbalance.plot(lw=3)\nresp_1.plot(lw=3)\nresp_2.plot(lw=3)\nresp_3.plot(lw=3)\nresp_4.plot(lw=3)\nplt.legend(loc=\"upper left\");\ndel resp_1\ndel resp_2\ndel resp_3\ndel resp_4\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-10-14T21:07:22.877252Z","iopub.execute_input":"2024-10-14T21:07:22.877643Z","iopub.status.idle":"2024-10-14T21:07:27.891944Z","shell.execute_reply.started":"2024-10-14T21:07:22.877607Z","shell.execute_reply":"2024-10-14T21:07:27.890874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Histogram of responder_0\n\nCarl histogram was a perfect balanced bell-shape. Mine is that \"Thing\". The Original Carl used 3000 Bins. I reduced to 30 to have a less messy figure. His distribution has very long tails. Mine is tailless:D\n\nOur values are between -0.05 and 0.05.","metadata":{}},{"cell_type":"code","source":"#By Carl McBride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\nplt.figure(figsize = (12,5))\nax = sns.distplot(train_data['responder_0'], \n             bins=30, #Original was 3000 bins\n             kde_kws={\"clip\":(-0.05,0.05)}, \n             hist_kws={\"range\":(-0.05,0.05)},\n             color='darkcyan', \n             kde=False);\nvalues = np.array([rec.get_height() for rec in ax.patches])\nnorm = plt.Normalize(values.min(), values.max())\ncolors = plt.cm.jet(norm(values))\nfor rec, col in zip(ax.patches, colors):\n    rec.set_color(col)\nplt.xlabel(\"Histogram of the resp values\", size=14)\nplt.show();\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-10-14T21:13:21.854835Z","iopub.execute_input":"2024-10-14T21:13:21.855805Z","iopub.status.idle":"2024-10-14T21:13:22.293811Z","shell.execute_reply.started":"2024-10-14T21:13:21.855759Z","shell.execute_reply":"2024-10-14T21:13:22.292784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"min_resp = train_data['responder_0'].min()\nprint('The minimum value for resp is: %.5f' % min_resp)\nmax_resp = train_data['responder_0'].max()\nprint('The maximum value for resp is:  %.5f' % max_resp)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T21:17:46.768637Z","iopub.execute_input":"2024-10-14T21:17:46.769581Z","iopub.status.idle":"2024-10-14T21:17:46.791206Z","shell.execute_reply.started":"2024-10-14T21:17:46.769540Z","shell.execute_reply":"2024-10-14T21:17:46.790100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Skew and Kurtosis of responder_0","metadata":{}},{"cell_type":"code","source":"min_resp = train_data['responder_0'].min()\nprint('The minimum value for resp is: %.5f' % min_resp)\nmax_resp = train_data['responder_0'].max()\nprint('The maximum value for resp is:  %.5f' % max_resp)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T21:18:32.745171Z","iopub.execute_input":"2024-10-14T21:18:32.745990Z","iopub.status.idle":"2024-10-14T21:18:32.766983Z","shell.execute_reply.started":"2024-10-14T21:18:32.745942Z","shell.execute_reply":"2024-10-14T21:18:32.765902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Fit a Cauchy distribution to this data\n\nEverything was 3000 below (1500/3000, initial_guess, bins) I reduced it to 30. Cause I trying to plot just a single responder_0\n\nProbably, you never saw before such \"Histogram\"","metadata":{}},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\n#I don't know why I had to define values again when I tried to change the numbers\n#Anyway this line is in every histogram\nvalues = np.array([rec.get_height() for rec in ax.patches])\n\nfrom scipy.optimize import curve_fit\n# the values\nx = list(range(len(values)))\nx = [((i)-15)/30 for i in x]\ny = values\n\ndef Lorentzian(x, x0, gamma, A):\n    return A * gamma**2/(gamma**2+( x - x0 )**2)\n\n# seed guess\ninitial_guess=(0, 0.001, 30)\n\n# the fit\nparameters,covariance=curve_fit(Lorentzian,x,y,initial_guess)\nsigma=np.sqrt(np.diag(covariance))\n\n# and plot\nplt.figure(figsize = (12,5))\nax = sns.distplot(train_data['responder_0'], \n             bins=30, \n             kde_kws={\"clip\":(-0.05,0.05)}, \n             hist_kws={\"range\":(-0.05,0.05)},\n             color='darkcyan', \n             kde=False);\nvalues = np.array([rec.get_height() for rec in ax.patches])\n#norm = plt.Normalize(values.min(), values.max())\n#colors = plt.cm.jet(norm(values))\n#for rec, col in zip(ax.patches, colors):\n#    rec.set_color(col)\nplt.xlabel(\"Histogram of the responder_0 values\", size=14)\nplt.plot(x,Lorentzian(x,*parameters),'--',color='black',lw=3)\nplt.show();\ndel values\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-10-14T21:43:31.713998Z","iopub.execute_input":"2024-10-14T21:43:31.715087Z","iopub.status.idle":"2024-10-14T21:43:32.189227Z","shell.execute_reply.started":"2024-10-14T21:43:31.715040Z","shell.execute_reply":"2024-10-14T21:43:32.188144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### No Missing values on weight on parquet (partition_id=4)\n\n\"Each trade has an associated weight and resp, which together represents a return on the trade.\"","metadata":{}},{"cell_type":"code","source":"percent_zeros = (100/train_data.shape[0])*((train_data.weight.values == 0).sum())\nprint('Percentage of zero weights is: %i' % percent_zeros +\"%\")","metadata":{"execution":{"iopub.status.busy":"2024-10-14T21:26:14.166545Z","iopub.execute_input":"2024-10-14T21:26:14.167126Z","iopub.status.idle":"2024-10-14T21:26:14.194179Z","shell.execute_reply.started":"2024-10-14T21:26:14.167079Z","shell.execute_reply":"2024-10-14T21:26:14.193192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Minimum and Maximum weight","metadata":{}},{"cell_type":"code","source":"min_weight = train_data['weight'].min()\nprint('The minimum weight is: %.2f' % min_weight)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T21:27:30.208868Z","iopub.execute_input":"2024-10-14T21:27:30.210779Z","iopub.status.idle":"2024-10-14T21:27:30.236339Z","shell.execute_reply.started":"2024-10-14T21:27:30.210722Z","shell.execute_reply":"2024-10-14T21:27:30.235006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_weight = train_data['weight'].max()\nprint('The maximum weight was: %.2f' % max_weight)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T21:30:21.099312Z","iopub.execute_input":"2024-10-14T21:30:21.100188Z","iopub.status.idle":"2024-10-14T21:30:21.117102Z","shell.execute_reply.started":"2024-10-14T21:30:21.100146Z","shell.execute_reply":"2024-10-14T21:30:21.116020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data[train_data['weight']==train_data['weight'].max()]","metadata":{"execution":{"iopub.status.busy":"2024-10-14T21:31:22.107338Z","iopub.execute_input":"2024-10-14T21:31:22.108235Z","iopub.status.idle":"2024-10-14T21:31:22.164519Z","shell.execute_reply.started":"2024-10-14T21:31:22.108190Z","shell.execute_reply":"2024-10-14T21:31:22.163385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Histogram of the non-zero weights\n\nThe original has 1400 bins. Our peak is situated at weight 1.24 Selling or buying?\n   ","metadata":{}},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\nplt.figure(figsize = (12,5))\nax = sns.distplot(train_data['weight'], \n             bins=140, #Original is 1400\n             kde_kws={\"clip\":(0.001,1.4)}, \n             hist_kws={\"range\":(0.001,1.4)},\n             color='darkcyan', \n             kde=False);\nvalues = np.array([rec.get_height() for rec in ax.patches])\nnorm = plt.Normalize(values.min(), values.max())\ncolors = plt.cm.jet(norm(values))\nfor rec, col in zip(ax.patches, colors):\n    rec.set_color(col)\nplt.xlabel(\"Histogram of non-zero weights\", size=14)\nplt.show();\ndel values\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-10-14T21:38:45.474360Z","iopub.execute_input":"2024-10-14T21:38:45.475162Z","iopub.status.idle":"2024-10-14T21:38:46.170393Z","shell.execute_reply.started":"2024-10-14T21:38:45.475116Z","shell.execute_reply":"2024-10-14T21:38:46.169149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## The logarithm of the non-zero weights (Histogram)","metadata":{}},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\ntrain_data_nonZero = train_data.query('weight > 0').reset_index(drop = True)\nplt.figure(figsize = (10,4))\nax = sns.distplot(np.log(train_data_nonZero['weight']), \n             bins=100, #Original was 1000\n             kde_kws={\"clip\":(-4,5)}, \n             hist_kws={\"range\":(-4,5)},\n             color='darkcyan', \n             kde=False);\nvalues = np.array([rec.get_height() for rec in ax.patches])\nnorm = plt.Normalize(values.min(), values.max())\ncolors = plt.cm.jet(norm(values))\nfor rec, col in zip(ax.patches, colors):\n    rec.set_color(col)\nplt.xlabel(\"Histogram of the logarithm of the non-zero weights\", size=14)\nplt.show();\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-10-14T21:37:36.186740Z","iopub.execute_input":"2024-10-14T21:37:36.187662Z","iopub.status.idle":"2024-10-14T21:37:38.675400Z","shell.execute_reply.started":"2024-10-14T21:37:36.187618Z","shell.execute_reply":"2024-10-14T21:37:38.674276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Try to fit a pair of Gaussian functions to this distribution\n\nI changed bins to 100 (Original was 1000) and (i/11)-4   to 11 (Original was 110)","metadata":{}},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\nvalues = np.array([rec.get_height() for rec in ax.patches])\n\nfrom scipy.optimize import curve_fit\n# the values\nx = list(range(len(values)))\nx = [(i/11)-4 for i in x]  #Original was 110\ny = values\n\n# define a Gaussian function\ndef Gaussian(x,mu,sigma,A):\n    return A*np.exp(-0.5 * ((x-mu)/sigma)**2)\n\ndef bimodal(x,mu_1,sigma_1,A_1,mu_2,sigma_2,A_2):\n    return Gaussian(x,mu_1,sigma_1,A_1) + Gaussian(x,mu_2,sigma_2,A_2)\n\n# seed guess\ninitial_guess=(1, 1 , 1,    1, 1, 1)\n\n# the fit\nparameters,covariance=curve_fit(bimodal,x,y,initial_guess)\nsigma=np.sqrt(np.diag(covariance))\n\n# the plot\nplt.figure(figsize = (10,4))\nax = sns.distplot(np.log(train_data_nonZero['weight']), \n             bins=100,     #Original was 1000\n             kde_kws={\"clip\":(-4,5)}, \n             hist_kws={\"range\":(-4,5)},\n             color='darkcyan', \n             kde=False);\nvalues = np.array([rec.get_height() for rec in ax.patches])\nnorm = plt.Normalize(values.min(), values.max())\ncolors = plt.cm.jet(norm(values))\nfor rec, col in zip(ax.patches, colors):\n    rec.set_color(col)\nplt.xlabel(\"Histogram of the logarithm of the non-zero weights\", size=14)\n# plot gaussian #1\nplt.plot(x,Gaussian(x,parameters[0],parameters[1],parameters[2]),':',color='black',lw=2,label='Gaussian #1', alpha=0.8)\n# plot gaussian #2\nplt.plot(x,Gaussian(x,parameters[3],parameters[4],parameters[5]),'--',color='black',lw=2,label='Gaussian #2', alpha=0.8)\n# plot the two gaussians together\nplt.plot(x,bimodal(x,*parameters),color='black',lw=2, alpha=0.7)\nplt.legend(loc=\"upper left\");\nplt.show();\ndel values\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-10-14T21:50:37.301570Z","iopub.execute_input":"2024-10-14T21:50:37.301977Z","iopub.status.idle":"2024-10-14T21:50:37.980289Z","shell.execute_reply.started":"2024-10-14T21:50:37.301941Z","shell.execute_reply":"2024-10-14T21:50:37.979152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Cumulative return\n\n In fact, I'm NOT sure if this chart is correct cause the features are a little bit different from the last competition. \n \n It was suppose to be \"**cumulative daily return over time**, which is given by weight multiplied by the value of responder\"","metadata":{}},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\ntrain_data['feature_00']   = train_data['weight']*train_data['responder_0']\ntrain_data['feature_01'] = train_data['weight']*train_data['responder_1']\ntrain_data['feature_02'] = train_data['weight']*train_data['responder_2']\ntrain_data['feature_03'] = train_data['weight']*train_data['responder_3']\ntrain_data['feature_04'] = train_data['weight']*train_data['responder_4']\n\nfig, ax = plt.subplots(figsize=(15, 5))\nresp    = pd.Series(1+(train_data.groupby('date_id')['feature_00'].mean())).cumprod()\nresp_1  = pd.Series(1+(train_data.groupby('date_id')['feature_01'].mean())).cumprod()\nresp_2  = pd.Series(1+(train_data.groupby('date_id')['feature_02'].mean())).cumprod()\nresp_3  = pd.Series(1+(train_data.groupby('date_id')['feature_03'].mean())).cumprod()\nresp_4  = pd.Series(1+(train_data.groupby('date_id')['feature_04'].mean())).cumprod()\nax.set_xlabel (\"Day\", fontsize=18)\nax.set_title (\"Cumulative daily return for responder and time horizons 1, 2, 3, and 4 (500 days)\", fontsize=18)\nresp.plot(lw=3, label='responder_0 x weight')\nresp_1.plot(lw=3, label='responder_1 x weight')\nresp_2.plot(lw=3, label='responder_2 x weight')\nresp_3.plot(lw=3, label='responder_3 x weight')\nresp_4.plot(lw=3, label='responder_4 x weight')\n# day 85 marker\nax.axvline(x=85, linestyle='--', alpha=0.3, c='red', lw=1)\nax.axvspan(0, 85 , color=sns.xkcd_rgb['grey'], alpha=0.1)\nplt.legend(loc=\"lower left\");","metadata":{"execution":{"iopub.status.busy":"2024-10-14T22:02:06.538503Z","iopub.execute_input":"2024-10-14T22:02:06.538918Z","iopub.status.idle":"2024-10-14T22:02:07.474220Z","shell.execute_reply.started":"2024-10-14T22:02:06.538876Z","shell.execute_reply":"2024-10-14T22:02:07.473171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"responder_0, responder_2 and responder_3, representing a more conservative strategy, result in the lowest return.\n\nIt seems that responder_1 has a more aggressive strategy (higher return).","metadata":{}},{"cell_type":"markdown","source":"## Histogram of the weight multiplied by the value of responder (after removing the 0 weights)","metadata":{}},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\ntrain_data_no_0 = train_data.query('weight > 0').reset_index(drop = True)\ntrain_data_no_0['wAbsResp'] = train_data_no_0['weight'] * (train_data_no_0['responder_0'])\n#plot\nplt.figure(figsize = (12,5))\nax = sns.distplot(train_data_no_0['wAbsResp'], \n             bins=150,  #Original was 1500\n             kde_kws={\"clip\":(-0.02,0.02)}, \n             hist_kws={\"range\":(-0.02,0.02)},\n             color='darkcyan', \n             kde=False);\nvalues = np.array([rec.get_height() for rec in ax.patches])\nnorm = plt.Normalize(values.min(), values.max())\ncolors = plt.cm.jet(norm(values))\nfor rec, col in zip(ax.patches, colors):\n    rec.set_color(col)\nplt.xlabel(\"Histogram of the weights * resp\", size=14)\nplt.show();","metadata":{"execution":{"iopub.status.busy":"2024-10-14T22:06:53.982379Z","iopub.execute_input":"2024-10-14T22:06:53.983311Z","iopub.status.idle":"2024-10-14T22:06:57.472943Z","shell.execute_reply.started":"2024-10-14T22:06:53.983267Z","shell.execute_reply":"2024-10-14T22:06:57.471944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Checking feature_00","metadata":{}},{"cell_type":"code","source":"train_data['feature_00'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-14T22:22:16.263913Z","iopub.execute_input":"2024-10-14T22:22:16.264346Z","iopub.status.idle":"2024-10-14T22:22:17.787780Z","shell.execute_reply.started":"2024-10-14T22:22:16.264307Z","shell.execute_reply":"2024-10-14T22:22:17.786733Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\nfig, ax = plt.subplots(figsize=(15, 4))\nfeature_0 = pd.Series(train_data['feature_00']).cumsum()\nax.set_xlabel (\"Trade\", fontsize=18)\nax.set_ylabel (\"feature_00 (cumulative)\", fontsize=18);\nfeature_0.plot(lw=3);","metadata":{"execution":{"iopub.status.busy":"2024-10-14T22:21:41.311360Z","iopub.execute_input":"2024-10-14T22:21:41.311736Z","iopub.status.idle":"2024-10-14T22:21:42.523448Z","shell.execute_reply.started":"2024-10-14T22:21:41.311702Z","shell.execute_reply":"2024-10-14T22:21:42.522401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### It seems to be four general 'types' of features, here is a plot of an example of one of each:\n\nIt was suppose to be linear/noisy/hybryd and negative. I couldn't find a linear feature. All seem to be hybryd. \n\nHybryds have both ascending and descending lines.","metadata":{}},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\nfig, ((ax1, ax2), (ax3, ax4)) = plt.subplots(2, 2,figsize=(20,10))\n\nax1.plot((pd.Series(train_data['feature_01']).cumsum()), lw=3, color='red')\nax1.set_title (\"Noisy\", fontsize=22);\nax1.axvline(x=514052, linestyle='--', alpha=0.3, c='green', lw=2)\nax1.axvspan(0, 514052 , color=sns.xkcd_rgb['grey'], alpha=0.1)\nax1.set_xlim(xmin=0)\nax1.set_ylabel (\"feature_01\", fontsize=18);\n\nax2.plot((pd.Series(train_data['feature_03']).cumsum()), lw=3, color='green')\nax2.set_title (\"Negative\", fontsize=22);\nax2.axvline(x=514052, linestyle='--', alpha=0.3, c='red', lw=2)\nax2.axvspan(0, 514052 , color=sns.xkcd_rgb['grey'], alpha=0.1)\nax2.set_xlim(xmin=0)\nax2.set_ylabel (\"feature_03\", fontsize=18);\n\nax3.plot((pd.Series(train_data['feature_55']).cumsum()), lw=3, color='darkorange')\nax3.set_title (\"Hybryd (Tag 21)\", fontsize=22);\nax3.set_xlabel (\"Trade\", fontsize=18)\nax3.axvline(x=514052, linestyle='--', alpha=0.3, c='green', lw=2)\nax3.axvspan(0, 514052 , color=sns.xkcd_rgb['grey'], alpha=0.1)\nax3.set_xlim(xmin=0)\nax3.set_ylabel (\"feature_55\", fontsize=18);\n\nax4.plot((pd.Series(train_data['feature_73']).cumsum()), lw=3, color='blue')\nax4.set_title (\"Hybryd\", fontsize=22)\nax4.set_xlabel (\"Trade\", fontsize=18)\nax4.set_ylabel (\"feature_73\", fontsize=18);\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-10-14T22:37:30.835347Z","iopub.execute_input":"2024-10-14T22:37:30.836752Z","iopub.status.idle":"2024-10-14T22:37:35.635706Z","shell.execute_reply.started":"2024-10-14T22:37:30.836675Z","shell.execute_reply":"2024-10-14T22:37:35.634654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Features 60 to 68 \n\nCarl chose that cause it was the tag 22 set.","metadata":{}},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\nfig, ax = plt.subplots(figsize=(15, 5))\nfeature_60= pd.Series(train_data['feature_60']).cumsum()\nfeature_61= pd.Series(train_data['feature_61']).cumsum()\nfeature_62= pd.Series(train_data['feature_62']).cumsum()\nfeature_63= pd.Series(train_data['feature_63']).cumsum()\nfeature_64= pd.Series(train_data['feature_64']).cumsum()\nfeature_65= pd.Series(train_data['feature_65']).cumsum()\nfeature_66= pd.Series(train_data['feature_66']).cumsum()\nfeature_67= pd.Series(train_data['feature_67']).cumsum()\nfeature_68= pd.Series(train_data['feature_68']).cumsum()\n#feature_69= pd.Series(train_data['feature_69']).cumsum()\nax.set_xlabel (\"Trade\", fontsize=18)\nax.set_title (\"Cumulative plot for feature_60 ... feature_68.\", fontsize=18)\nfeature_60.plot(lw=3)\nfeature_61.plot(lw=3)\nfeature_62.plot(lw=3)\nfeature_63.plot(lw=3)\nfeature_64.plot(lw=3)\nfeature_65.plot(lw=3)\nfeature_66.plot(lw=3)\nfeature_67.plot(lw=3)\nfeature_68.plot(lw=3)\n#feature_69.plot(lw=3)\nplt.legend(loc=\"upper left\");\ndel feature_60, feature_61, feature_62, feature_63, feature_64, feature_65, feature_66 ,feature_67, feature_68\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-10-14T22:40:58.272602Z","iopub.execute_input":"2024-10-14T22:40:58.273494Z","iopub.status.idle":"2024-10-14T22:41:06.492366Z","shell.execute_reply.started":"2024-10-14T22:40:58.273446Z","shell.execute_reply":"2024-10-14T22:41:06.491171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Trying to check some distributions","metadata":{}},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\nsns.set_palette(\"bright\")\n\nfig, axes = plt.subplots(2,2,figsize=(8,8))\n\nsns.distplot(train_data[['feature_60']], hist=True, bins=200,  ax=axes[0,0])\nsns.distplot(train_data[['feature_61']], hist=True, bins=200,  ax=axes[0,0])\naxes[0,0].set_title (\"features 60 and 61\", fontsize=18)\naxes[0,0].legend(labels=['60', '61'])\n\nsns.distplot(train_data[['feature_62']], hist=True,  bins=200, ax=axes[0,1])\nsns.distplot(train_data[['feature_63']], hist=True,  bins=200, ax=axes[0,1])\naxes[0,1].set_title (\"features 62 and 63\", fontsize=18)\naxes[0,1].legend(labels=['62', '63'])\n\nsns.distplot(train_data[['feature_65']], hist=True,  bins=200, ax=axes[1,0])\nsns.distplot(train_data[['feature_66']], hist=True,  bins=200, ax=axes[1,0])\naxes[1,0].set_title (\"features 65 and 66\", fontsize=18)\naxes[1,0].legend(labels=['65', '66'])\n\n\nsns.distplot(train_data[['feature_67']], hist=True,  bins=200, ax=axes[1,1])\nsns.distplot(train_data[['feature_68']], hist=True,  bins=200, ax=axes[1,1])\naxes[1,1].set_title (\"features 67 and 68\", fontsize=18)\naxes[1,1].legend(labels=['67', '68'])\n\nplt.show();\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-10-14T22:42:03.401130Z","iopub.execute_input":"2024-10-14T22:42:03.401614Z","iopub.status.idle":"2024-10-14T22:44:42.295413Z","shell.execute_reply.started":"2024-10-14T22:42:03.401574Z","shell.execute_reply":"2024-10-14T22:44:42.294238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature_64 histogram\n\nMaybe on the last competition it was distinct from the other. Not now.","metadata":{}},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\nplt.figure(figsize = (12,5))\nax = sns.distplot(train_data['feature_64'], \n             bins=120, \n             kde_kws={\"clip\":(-6,6)}, \n             hist_kws={\"range\":(-6,6)},\n             color='darkcyan', \n             kde=False);\nvalues = np.array([rec.get_height() for rec in ax.patches])\nnorm = plt.Normalize(values.min(), values.max())\ncolors = plt.cm.jet(norm(values))\nfor rec, col in zip(ax.patches, colors):\n    rec.set_color(col)\nplt.xlabel(\"Histogram of feature_64\", size=14)\nplt.show();\ndel values\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-10-14T22:45:18.291070Z","iopub.execute_input":"2024-10-14T22:45:18.292083Z","iopub.status.idle":"2024-10-14T22:45:19.054213Z","shell.execute_reply.started":"2024-10-14T22:45:18.292023Z","shell.execute_reply":"2024-10-14T22:45:19.053006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##  Plot of feature_51 w.r.t. weight for non-zero weights:\n\nIt seems a wav file.","metadata":{}},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\nfig, ax = plt.subplots(figsize=(15, 4))\nax.scatter(train_data_nonZero.weight, train_data_nonZero.feature_51, s=0.1, color='b')\nax.set_xlabel('weight')\nax.set_ylabel('feature_51')\nplt.show();","metadata":{"execution":{"iopub.status.busy":"2024-10-14T22:47:28.881539Z","iopub.execute_input":"2024-10-14T22:47:28.882659Z","iopub.status.idle":"2024-10-14T22:47:31.151833Z","shell.execute_reply.started":"2024-10-14T22:47:28.882599Z","shell.execute_reply":"2024-10-14T22:47:31.150796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\nfig, ax = plt.subplots(figsize=(15, 5))\nfeature_55= pd.Series(train_data['feature_55']).cumsum()\nfeature_56= pd.Series(train_data['feature_56']).cumsum()\nfeature_57= pd.Series(train_data['feature_57']).cumsum()\nfeature_58= pd.Series(train_data['feature_58']).cumsum()\nfeature_59= pd.Series(train_data['feature_59']).cumsum()\nax.set_xlabel (\"Trade\", fontsize=18)\nax.set_title (\"Cumulative plot for the 'Tag 21' features (55-59)\", fontsize=18)\nax.axvline(x=514052, linestyle='--', alpha=0.3, c='black', lw=1)\nax.axvspan(0,  514052, color=sns.xkcd_rgb['grey'], alpha=0.1)\nfeature_55.plot(lw=3)\nfeature_56.plot(lw=3)\nfeature_57.plot(lw=3)\nfeature_58.plot(lw=3)\nfeature_59.plot(lw=3)\nplt.legend(loc=\"upper left\");\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-10-14T22:49:19.127301Z","iopub.execute_input":"2024-10-14T22:49:19.128352Z","iopub.status.idle":"2024-10-14T22:49:24.171776Z","shell.execute_reply.started":"2024-10-14T22:49:19.128298Z","shell.execute_reply":"2024-10-14T22:49:24.170758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I tried to make a RandomForestRegressor though I got stucked on max_features, then the Errors only changed more an more. Not funny.\n\nInvalidParameterError: The 'max_features' parameter of RandomForestRegressor must be an int in the range [1, inf), a float in the range (0.0, 1.0], a str among {'log2', 'sqrt'} or None. Got 'auto' instead.","metadata":{}},{"cell_type":"markdown","source":"#On the next Competition, Jane Street should be changed to Jane Avenue, due to its huge data. And huge company.\n\n![](https://encrypted-tbn0.gstatic.com/images?q=tbn:ANd9GcSUvZbo9f3xkzvZqCiW_YVCgo6ESFUUqqlV4w&s)\nwikipedia","metadata":{}},{"cell_type":"markdown","source":"#Acknowledgements:\n\nCarl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook","metadata":{"execution":{"iopub.status.busy":"2024-10-14T23:43:24.887998Z","iopub.execute_input":"2024-10-14T23:43:24.888897Z","iopub.status.idle":"2024-10-14T23:43:25.039417Z","shell.execute_reply.started":"2024-10-14T23:43:24.888848Z","shell.execute_reply":"2024-10-14T23:43:25.038067Z"}}}]}