{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2025-01-07T16:13:21.324026Z","iopub.execute_input":"2025-01-07T16:13:21.324498Z","iopub.status.idle":"2025-01-07T16:13:22.820729Z","shell.execute_reply.started":"2025-01-07T16:13:21.324436Z","shell.execute_reply":"2025-01-07T16:13:22.819373Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# plotting stuff\nfrom pandas.plotting import lag_plot\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nimport plotly.graph_objects as go\ncolorMap = sns.light_palette(\"blue\", as_cmap=True)\n\n# system\nimport warnings\nwarnings.filterwarnings('ignore')\n# for the image import\n\nfrom IPython.display import Image\n# garbage collector to keep RAM in check\nimport gc \n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:22.823583Z","iopub.execute_input":"2025-01-07T16:13:22.824073Z","iopub.status.idle":"2025-01-07T16:13:25.530748Z","shell.execute_reply.started":"2025-01-07T16:13:22.824003Z","shell.execute_reply":"2025-01-07T16:13:25.529673Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import SimpleImputer\nfrom lightgbm import LGBMRegressor\nfrom sklearn.metrics import mean_squared_error","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:25.531924Z","iopub.execute_input":"2025-01-07T16:13:25.532413Z","iopub.status.idle":"2025-01-07T16:13:27.196898Z","shell.execute_reply.started":"2025-01-07T16:13:25.532380Z","shell.execute_reply":"2025-01-07T16:13:27.195355Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"resp = pd.read_csv('/kaggle/input/jane-street-real-time-market-data-forecasting/responders.csv')\nresp.tail()","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:27.198186Z","iopub.execute_input":"2025-01-07T16:13:27.198794Z","iopub.status.idle":"2025-01-07T16:13:27.235021Z","shell.execute_reply.started":"2025-01-07T16:13:27.198759Z","shell.execute_reply":"2025-01-07T16:13:27.233891Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Features file, index_col=0 so the index is feature column\n\nfeat = pd.read_csv('/kaggle/input/jane-street-real-time-market-data-forecasting/features.csv', index_col=0)\nfeat.tail()","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:27.236333Z","iopub.execute_input":"2025-01-07T16:13:27.236697Z","iopub.status.idle":"2025-01-07T16:13:27.267784Z","shell.execute_reply.started":"2025-01-07T16:13:27.236662Z","shell.execute_reply":"2025-01-07T16:13:27.266010Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Features file, index_col=0 so the index is feature column\n\nfeat = pd.read_csv('/kaggle/input/jane-street-real-time-market-data-forecasting/features.csv', index_col=0)\nfeat.tail()","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:27.269767Z","iopub.execute_input":"2025-01-07T16:13:27.272117Z","iopub.status.idle":"2025-01-07T16:13:27.297269Z","shell.execute_reply.started":"2025-01-07T16:13:27.272077Z","shell.execute_reply":"2025-01-07T16:13:27.295926Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#By Safadoost\n#encode the boolean  True/False to plot a heatmap after\n\n#By Safadoost\n#encode the boolean  True/False to plot a heatmap after\n\ncategorical_columns = feat[:]\n# Create a new column for each unique value in the categorical columns.\ndf_categorical = pd.get_dummies(\n    data = feat ,\n    prefix = 'OHE' ,  # One hot encoding\n    prefix_sep = '_' ,\n    columns = categorical_columns ,\n    #drop_first = True ,\n    dtype = 'int8'\n)\n","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:27.302217Z","iopub.execute_input":"2025-01-07T16:13:27.302859Z","iopub.status.idle":"2025-01-07T16:13:27.338475Z","shell.execute_reply.started":"2025-01-07T16:13:27.302800Z","shell.execute_reply":"2025-01-07T16:13:27.337263Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12,10))\nsns.heatmap(df_categorical.corr(),annot=True);","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:27.339956Z","iopub.execute_input":"2025-01-07T16:13:27.340380Z","iopub.status.idle":"2025-01-07T16:13:28.629001Z","shell.execute_reply.started":"2025-01-07T16:13:27.340335Z","shell.execute_reply":"2025-01-07T16:13:28.627450Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Read One parquet file. Obviously, it's big.\n\ntrain_data = pd.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=4/part-0.parquet\")\ntrain_data.tail()","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:28.631030Z","iopub.execute_input":"2025-01-07T16:13:28.631586Z","iopub.status.idle":"2025-01-07T16:13:40.696454Z","shell.execute_reply.started":"2025-01-07T16:13:28.631494Z","shell.execute_reply":"2025-01-07T16:13:40.695216Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#By Carl McBride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\nfig, ax = plt.subplots(figsize=(15, 5))\nbalance= pd.Series(train_data['responder_0']).cumsum()\nax.set_xlabel (\"Trade\", fontsize=18)\nax.set_ylabel (\"Cumulative responder 0\", fontsize=18);\nbalance.plot(lw=3);\ndel balance\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:40.697907Z","iopub.execute_input":"2025-01-07T16:13:40.698248Z","iopub.status.idle":"2025-01-07T16:13:42.216312Z","shell.execute_reply.started":"2025-01-07T16:13:40.698215Z","shell.execute_reply":"2025-01-07T16:13:42.214816Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#By Carl McBride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\nfig, ax = plt.subplots(figsize=(15, 5))\nbalance= pd.Series(train_data['responder_0']).cumsum()\nresp_1= pd.Series(train_data['responder_1']).cumsum()\nresp_2= pd.Series(train_data['responder_2']).cumsum()\nresp_3= pd.Series(train_data['responder_3']).cumsum()\nresp_4= pd.Series(train_data['responder_4']).cumsum()\nax.set_xlabel (\"Trade\", fontsize=18)\nax.set_title (\"Cumulative resp and time horizons 1, 2, 3, and 4 (500 days)\", fontsize=18)\nbalance.plot(lw=3)\nresp_1.plot(lw=3)\nresp_2.plot(lw=3)\nresp_3.plot(lw=3)\nresp_4.plot(lw=3)\nplt.legend(loc=\"upper left\");\ndel resp_1\ndel resp_2\ndel resp_3\ndel resp_4\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:42.217814Z","iopub.execute_input":"2025-01-07T16:13:42.218176Z","iopub.status.idle":"2025-01-07T16:13:47.601341Z","shell.execute_reply.started":"2025-01-07T16:13:42.218139Z","shell.execute_reply":"2025-01-07T16:13:47.600111Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#By Carl McBride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\nplt.figure(figsize = (12,5))\nax = sns.distplot(train_data['responder_0'], \n             bins=30, #Original was 3000 bins\n             kde_kws={\"clip\":(-0.05,0.05)}, \n             hist_kws={\"range\":(-0.05,0.05)},\n             color='darkcyan', \n             kde=False);\nvalues = np.array([rec.get_height() for rec in ax.patches])\nnorm = plt.Normalize(values.min(), values.max())\ncolors = plt.cm.jet(norm(values))\nfor rec, col in zip(ax.patches, colors):\n    rec.set_color(col)\nplt.xlabel(\"Histogram of the resp values\", size=14)\nplt.show();\ngc.collect();\n","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:47.602847Z","iopub.execute_input":"2025-01-07T16:13:47.603259Z","iopub.status.idle":"2025-01-07T16:13:48.006987Z","shell.execute_reply.started":"2025-01-07T16:13:47.603215Z","shell.execute_reply":"2025-01-07T16:13:48.005596Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"min_resp = train_data['responder_0'].min()\nprint('The minimum value for resp is: %.5f' % min_resp)\nmax_resp = train_data['responder_0'].max()\nprint('The maximum value for resp is:  %.5f' % max_resp)","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:48.008330Z","iopub.execute_input":"2025-01-07T16:13:48.008678Z","iopub.status.idle":"2025-01-07T16:13:48.036605Z","shell.execute_reply.started":"2025-01-07T16:13:48.008645Z","shell.execute_reply":"2025-01-07T16:13:48.034109Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"min_resp = train_data['responder_0'].min()\nprint('The minimum value for resp is: %.5f' % min_resp)\nmax_resp = train_data['responder_0'].max()\nprint('The maximum value for resp is:  %.5f' % max_resp)","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:48.038500Z","iopub.execute_input":"2025-01-07T16:13:48.039042Z","iopub.status.idle":"2025-01-07T16:13:48.064301Z","shell.execute_reply.started":"2025-01-07T16:13:48.038993Z","shell.execute_reply":"2025-01-07T16:13:48.062960Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"percent_zeros = (100/train_data.shape[0])*((train_data.weight.values == 0).sum())\nprint('Percentage of zero weights is: %i' % percent_zeros +\"%\")","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:48.066076Z","iopub.execute_input":"2025-01-07T16:13:48.066587Z","iopub.status.idle":"2025-01-07T16:13:48.079530Z","shell.execute_reply.started":"2025-01-07T16:13:48.066537Z","shell.execute_reply":"2025-01-07T16:13:48.078233Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"min_weight = train_data['weight'].min()\nprint('The minimum weight is: %.2f' % min_weight)","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:48.081205Z","iopub.execute_input":"2025-01-07T16:13:48.081573Z","iopub.status.idle":"2025-01-07T16:13:48.099240Z","shell.execute_reply.started":"2025-01-07T16:13:48.081529Z","shell.execute_reply":"2025-01-07T16:13:48.097928Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"max_weight = train_data['weight'].max()\nprint('The maximum weight was: %.2f' % max_weight)","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:48.101140Z","iopub.execute_input":"2025-01-07T16:13:48.101440Z","iopub.status.idle":"2025-01-07T16:13:48.119168Z","shell.execute_reply.started":"2025-01-07T16:13:48.101410Z","shell.execute_reply":"2025-01-07T16:13:48.117866Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data[train_data['weight']==train_data['weight'].max()]","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:48.120717Z","iopub.execute_input":"2025-01-07T16:13:48.121209Z","iopub.status.idle":"2025-01-07T16:13:48.164124Z","shell.execute_reply.started":"2025-01-07T16:13:48.121156Z","shell.execute_reply":"2025-01-07T16:13:48.162894Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\nplt.figure(figsize = (12,5))\nax = sns.distplot(train_data['weight'], \n             bins=140, #Original is 1400\n             kde_kws={\"clip\":(0.001,1.4)}, \n             hist_kws={\"range\":(0.001,1.4)},\n             color='darkcyan', \n             kde=False);\nvalues = np.array([rec.get_height() for rec in ax.patches])\nnorm = plt.Normalize(values.min(), values.max())\ncolors = plt.cm.jet(norm(values))\nfor rec, col in zip(ax.patches, colors):\n    rec.set_color(col)\nplt.xlabel(\"Histogram of non-zero weights\", size=14)\nplt.show();\ndel values\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:48.165426Z","iopub.execute_input":"2025-01-07T16:13:48.165933Z","iopub.status.idle":"2025-01-07T16:13:48.759232Z","shell.execute_reply.started":"2025-01-07T16:13:48.165846Z","shell.execute_reply":"2025-01-07T16:13:48.758076Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\ntrain_data_nonZero = train_data.query('weight > 0').reset_index(drop = True)\nplt.figure(figsize = (10,4))\nax = sns.distplot(np.log(train_data_nonZero['weight']), \n             bins=100, #Original was 1000\n             kde_kws={\"clip\":(-4,5)}, \n             hist_kws={\"range\":(-4,5)},\n             color='darkcyan', \n             kde=False);\nvalues = np.array([rec.get_height() for rec in ax.patches])\nnorm = plt.Normalize(values.min(), values.max())\ncolors = plt.cm.jet(norm(values))\nfor rec, col in zip(ax.patches, colors):\n    rec.set_color(col)\nplt.xlabel(\"Histogram of the logarithm of the non-zero weights\", size=14)\nplt.show();\ngc.collect();\n","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:48.760803Z","iopub.execute_input":"2025-01-07T16:13:48.761191Z","iopub.status.idle":"2025-01-07T16:13:51.531119Z","shell.execute_reply.started":"2025-01-07T16:13:48.761155Z","shell.execute_reply":"2025-01-07T16:13:51.529887Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\nvalues = np.array([rec.get_height() for rec in ax.patches])\n\nfrom scipy.optimize import curve_fit\n# the values\nx = list(range(len(values)))\nx = [(i/11)-4 for i in x]  #Original was 110\ny = values\n\n# define a Gaussian function\ndef Gaussian(x,mu,sigma,A):\n    return A*np.exp(-0.5 * ((x-mu)/sigma)**2)\n\ndef bimodal(x,mu_1,sigma_1,A_1,mu_2,sigma_2,A_2):\n    return Gaussian(x,mu_1,sigma_1,A_1) + Gaussian(x,mu_2,sigma_2,A_2)\n\n# seed guess\ninitial_guess=(1, 1 , 1,    1, 1, 1)\n\n# the fit\nparameters,covariance=curve_fit(bimodal,x,y,initial_guess)\nsigma=np.sqrt(np.diag(covariance))\n\n# the plot\nplt.figure(figsize = (10,4))\nax = sns.distplot(np.log(train_data_nonZero['weight']), \n             bins=100,     #Original was 1000\n             kde_kws={\"clip\":(-4,5)}, \n             hist_kws={\"range\":(-4,5)},\n             color='darkcyan', \n             kde=False);\nvalues = np.array([rec.get_height() for rec in ax.patches])\nnorm = plt.Normalize(values.min(), values.max())\ncolors = plt.cm.jet(norm(values))\nfor rec, col in zip(ax.patches, colors):\n    rec.set_color(col)\nplt.xlabel(\"Histogram of the logarithm of the non-zero weights\", size=14)\n# plot gaussian #1\nplt.plot(x,Gaussian(x,parameters[0],parameters[1],parameters[2]),':',color='black',lw=2,label='Gaussian #1', alpha=0.8)\n# plot gaussian #2\nplt.plot(x,Gaussian(x,parameters[3],parameters[4],parameters[5]),'--',color='black',lw=2,label='Gaussian #2', alpha=0.8)\n# plot the two gaussians together\nplt.plot(x,bimodal(x,*parameters),color='black',lw=2, alpha=0.7)\nplt.legend(loc=\"upper left\");\nplt.show();\ndel values\ngc.collect();\n","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:51.532457Z","iopub.execute_input":"2025-01-07T16:13:51.532996Z","iopub.status.idle":"2025-01-07T16:13:52.198874Z","shell.execute_reply.started":"2025-01-07T16:13:51.532960Z","shell.execute_reply":"2025-01-07T16:13:52.197547Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\ntrain_data['feature_00']   = train_data['weight']*train_data['responder_0']\ntrain_data['feature_01'] = train_data['weight']*train_data['responder_1']\ntrain_data['feature_02'] = train_data['weight']*train_data['responder_2']\ntrain_data['feature_03'] = train_data['weight']*train_data['responder_3']\ntrain_data['feature_04'] = train_data['weight']*train_data['responder_4']\n\nfig, ax = plt.subplots(figsize=(15, 5))\nresp    = pd.Series(1+(train_data.groupby('date_id')['feature_00'].mean())).cumprod()\nresp_1  = pd.Series(1+(train_data.groupby('date_id')['feature_01'].mean())).cumprod()\nresp_2  = pd.Series(1+(train_data.groupby('date_id')['feature_02'].mean())).cumprod()\nresp_3  = pd.Series(1+(train_data.groupby('date_id')['feature_03'].mean())).cumprod()\nresp_4  = pd.Series(1+(train_data.groupby('date_id')['feature_04'].mean())).cumprod()\nax.set_xlabel (\"Day\", fontsize=18)\nax.set_title (\"Cumulative daily return for responder and time horizons 1, 2, 3, and 4 (500 days)\", fontsize=18)\nresp.plot(lw=3, label='responder_0 x weight')\nresp_1.plot(lw=3, label='responder_1 x weight')\nresp_2.plot(lw=3, label='responder_2 x weight')\nresp_3.plot(lw=3, label='responder_3 x weight')\nresp_4.plot(lw=3, label='responder_4 x weight')\n# day 85 marker\nax.axvline(x=85, linestyle='--', alpha=0.3, c='red', lw=1)\nax.axvspan(0, 85 , color=sns.xkcd_rgb['grey'], alpha=0.1)\nplt.legend(loc=\"lower left\");\n","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:52.206448Z","iopub.execute_input":"2025-01-07T16:13:52.206851Z","iopub.status.idle":"2025-01-07T16:13:53.304080Z","shell.execute_reply.started":"2025-01-07T16:13:52.206808Z","shell.execute_reply":"2025-01-07T16:13:53.302815Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\ntrain_data_no_0 = train_data.query('weight > 0').reset_index(drop = True)\ntrain_data_no_0['wAbsResp'] = train_data_no_0['weight'] * (train_data_no_0['responder_0'])\n#plot\nplt.figure(figsize = (12,5))\nax = sns.distplot(train_data_no_0['wAbsResp'], \n             bins=150,  #Original was 1500\n             kde_kws={\"clip\":(-0.02,0.02)}, \n             hist_kws={\"range\":(-0.02,0.02)},\n             color='darkcyan', \n             kde=False);\nvalues = np.array([rec.get_height() for rec in ax.patches])\nnorm = plt.Normalize(values.min(), values.max())\ncolors = plt.cm.jet(norm(values))\nfor rec, col in zip(ax.patches, colors):\n    rec.set_color(col)\nplt.xlabel(\"Histogram of the weights * resp\", size=14)\nplt.show();","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:53.305611Z","iopub.execute_input":"2025-01-07T16:13:53.306247Z","iopub.status.idle":"2025-01-07T16:13:57.532470Z","shell.execute_reply.started":"2025-01-07T16:13:53.306197Z","shell.execute_reply":"2025-01-07T16:13:57.531283Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data['feature_00'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:57.534035Z","iopub.execute_input":"2025-01-07T16:13:57.534488Z","iopub.status.idle":"2025-01-07T16:13:58.536252Z","shell.execute_reply.started":"2025-01-07T16:13:57.534441Z","shell.execute_reply":"2025-01-07T16:13:58.534828Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\nfig, ax = plt.subplots(figsize=(15, 4))\nfeature_0 = pd.Series(train_data['feature_00']).cumsum()\nax.set_xlabel (\"Trade\", fontsize=18)\nax.set_ylabel (\"feature_00 (cumulative)\", fontsize=18);\nfeature_0.plot(lw=3);","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:58.537685Z","iopub.execute_input":"2025-01-07T16:13:58.538039Z","iopub.status.idle":"2025-01-07T16:13:59.751233Z","shell.execute_reply.started":"2025-01-07T16:13:58.538006Z","shell.execute_reply":"2025-01-07T16:13:59.750041Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\nfig, ((ax1, ax2), (ax3, ax4)) = plt.subplots(2, 2,figsize=(20,10))\n\nax1.plot((pd.Series(train_data['feature_01']).cumsum()), lw=3, color='red')\nax1.set_title (\"Noisy\", fontsize=22);\nax1.axvline(x=514052, linestyle='--', alpha=0.3, c='green', lw=2)\nax1.axvspan(0, 514052 , color=sns.xkcd_rgb['grey'], alpha=0.1)\nax1.set_xlim(xmin=0)\nax1.set_ylabel (\"feature_01\", fontsize=18);\n\nax2.plot((pd.Series(train_data['feature_03']).cumsum()), lw=3, color='green')\nax2.set_title (\"Negative\", fontsize=22);\nax2.axvline(x=514052, linestyle='--', alpha=0.3, c='red', lw=2)\nax2.axvspan(0, 514052 , color=sns.xkcd_rgb['grey'], alpha=0.1)\nax2.set_xlim(xmin=0)\nax2.set_ylabel (\"feature_03\", fontsize=18);\n\nax3.plot((pd.Series(train_data['feature_55']).cumsum()), lw=3, color='darkorange')\nax3.set_title (\"Hybryd (Tag 21)\", fontsize=22);\nax3.set_xlabel (\"Trade\", fontsize=18)\nax3.axvline(x=514052, linestyle='--', alpha=0.3, c='green', lw=2)\nax3.axvspan(0, 514052 , color=sns.xkcd_rgb['grey'], alpha=0.1)\nax3.set_xlim(xmin=0)\nax3.set_ylabel (\"feature_55\", fontsize=18);\n\nax4.plot((pd.Series(train_data['feature_73']).cumsum()), lw=3, color='blue')\nax4.set_title (\"Hybryd\", fontsize=22)\nax4.set_xlabel (\"Trade\", fontsize=18)\nax4.set_ylabel (\"feature_73\", fontsize=18);\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:13:59.752555Z","iopub.execute_input":"2025-01-07T16:13:59.752867Z","iopub.status.idle":"2025-01-07T16:14:04.312856Z","shell.execute_reply.started":"2025-01-07T16:13:59.752837Z","shell.execute_reply":"2025-01-07T16:14:04.311582Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\nfig, ax = plt.subplots(figsize=(15, 5))\nfeature_60= pd.Series(train_data['feature_60']).cumsum()\nfeature_61= pd.Series(train_data['feature_61']).cumsum()\nfeature_62= pd.Series(train_data['feature_62']).cumsum()\nfeature_63= pd.Series(train_data['feature_63']).cumsum()\nfeature_64= pd.Series(train_data['feature_64']).cumsum()\nfeature_65= pd.Series(train_data['feature_65']).cumsum()\nfeature_66= pd.Series(train_data['feature_66']).cumsum()\nfeature_67= pd.Series(train_data['feature_67']).cumsum()\nfeature_68= pd.Series(train_data['feature_68']).cumsum()\n#feature_69= pd.Series(train_data['feature_69']).cumsum()\nax.set_xlabel (\"Trade\", fontsize=18)\nax.set_title (\"Cumulative plot for feature_60 ... feature_68.\", fontsize=18)\nfeature_60.plot(lw=3)\nfeature_61.plot(lw=3)\nfeature_62.plot(lw=3)\nfeature_63.plot(lw=3)\nfeature_64.plot(lw=3)\nfeature_65.plot(lw=3)\nfeature_66.plot(lw=3)\nfeature_67.plot(lw=3)\nfeature_68.plot(lw=3)\n#feature_69.plot(lw=3)\nplt.legend(loc=\"upper left\");\ndel feature_60, feature_61, feature_62, feature_63, feature_64, feature_65, feature_66 ,feature_67, feature_68\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:14:04.314414Z","iopub.execute_input":"2025-01-07T16:14:04.314907Z","iopub.status.idle":"2025-01-07T16:14:12.799406Z","shell.execute_reply.started":"2025-01-07T16:14:04.314846Z","shell.execute_reply":"2025-01-07T16:14:12.798305Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\nsns.set_palette(\"bright\")\n\nfig, axes = plt.subplots(2,2,figsize=(8,8))\n\nsns.distplot(train_data[['feature_60']], hist=True, bins=200,  ax=axes[0,0])\nsns.distplot(train_data[['feature_61']], hist=True, bins=200,  ax=axes[0,0])\naxes[0,0].set_title (\"features 60 and 61\", fontsize=18)\naxes[0,0].legend(labels=['60', '61'])\n\nsns.distplot(train_data[['feature_62']], hist=True,  bins=200, ax=axes[0,1])\nsns.distplot(train_data[['feature_63']], hist=True,  bins=200, ax=axes[0,1])\naxes[0,1].set_title (\"features 62 and 63\", fontsize=18)\naxes[0,1].legend(labels=['62', '63'])\n\nsns.distplot(train_data[['feature_65']], hist=True,  bins=200, ax=axes[1,0])\nsns.distplot(train_data[['feature_66']], hist=True,  bins=200, ax=axes[1,0])\naxes[1,0].set_title (\"features 65 and 66\", fontsize=18)\naxes[1,0].legend(labels=['65', '66'])\n\n\nsns.distplot(train_data[['feature_67']], hist=True,  bins=200, ax=axes[1,1])\nsns.distplot(train_data[['feature_68']], hist=True,  bins=200, ax=axes[1,1])\naxes[1,1].set_title (\"features 67 and 68\", fontsize=18)\naxes[1,1].legend(labels=['67', '68'])\n\nplt.show();\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:14:12.800732Z","iopub.execute_input":"2025-01-07T16:14:12.801081Z","iopub.status.idle":"2025-01-07T16:16:59.749743Z","shell.execute_reply.started":"2025-01-07T16:14:12.801048Z","shell.execute_reply":"2025-01-07T16:16:59.748579Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\nplt.figure(figsize = (12,5))\nax = sns.distplot(train_data['feature_64'], \n             bins=120, \n             kde_kws={\"clip\":(-6,6)}, \n             hist_kws={\"range\":(-6,6)},\n             color='darkcyan', \n             kde=False);\nvalues = np.array([rec.get_height() for rec in ax.patches])\nnorm = plt.Normalize(values.min(), values.max())\ncolors = plt.cm.jet(norm(values))\nfor rec, col in zip(ax.patches, colors):\n    rec.set_color(col)\nplt.xlabel(\"Histogram of feature_64\", size=14)\nplt.show();\ndel values\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:16:59.751225Z","iopub.execute_input":"2025-01-07T16:16:59.751702Z","iopub.status.idle":"2025-01-07T16:17:00.471299Z","shell.execute_reply.started":"2025-01-07T16:16:59.751655Z","shell.execute_reply":"2025-01-07T16:17:00.470087Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\nfig, ax = plt.subplots(figsize=(15, 4))\nax.scatter(train_data_nonZero.weight, train_data_nonZero.feature_51, s=0.1, color='b')\nax.set_xlabel('weight')\nax.set_ylabel('feature_51')\nplt.show();","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:17:00.473072Z","iopub.execute_input":"2025-01-07T16:17:00.473423Z","iopub.status.idle":"2025-01-07T16:17:02.806369Z","shell.execute_reply.started":"2025-01-07T16:17:00.473389Z","shell.execute_reply":"2025-01-07T16:17:02.805085Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#By Carl Mcbride Ellis https://www.kaggle.com/code/carlmcbrideellis/jane-street-eda-of-day-0-and-feature-importance/notebook\n\nfig, ax = plt.subplots(figsize=(15, 5))\nfeature_55= pd.Series(train_data['feature_55']).cumsum()\nfeature_56= pd.Series(train_data['feature_56']).cumsum()\nfeature_57= pd.Series(train_data['feature_57']).cumsum()\nfeature_58= pd.Series(train_data['feature_58']).cumsum()\nfeature_59= pd.Series(train_data['feature_59']).cumsum()\nax.set_xlabel (\"Trade\", fontsize=18)\nax.set_title (\"Cumulative plot for the 'Tag 21' features (55-59)\", fontsize=18)\nax.axvline(x=514052, linestyle='--', alpha=0.3, c='black', lw=1)\nax.axvspan(0,  514052, color=sns.xkcd_rgb['grey'], alpha=0.1)\nfeature_55.plot(lw=3)\nfeature_56.plot(lw=3)\nfeature_57.plot(lw=3)\nfeature_58.plot(lw=3)\nfeature_59.plot(lw=3)\nplt.legend(loc=\"upper left\");\ngc.collect();\n","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:17:02.807979Z","iopub.execute_input":"2025-01-07T16:17:02.808395Z","iopub.status.idle":"2025-01-07T16:17:07.943192Z","shell.execute_reply.started":"2025-01-07T16:17:02.808350Z","shell.execute_reply":"2025-01-07T16:17:07.941769Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom itertools import combinations, product\n\ndata = np.array([\n    [1, 0, 1, 0, 0],\n    [1, 0, 0, 1, 0],\n    [1, 1, 0, 0, 0],\n    [0, 0, 1, 0, 1],\n    [0, 0, 0, 1, 1],\n    [0, 1, 0, 0, 1],\n    [0, 0, 1, 0, 0],\n    [0, 0, 0, 1, 0],\n    [0, 1, 0, 0, 0]\n])\n\ndf = pd.DataFrame(data.T, columns=[f'r{i}' for i in range(9)])\n\ncandidates = [1, 2, 4, 5, 7, 8]  # r1, r2, r4, r5, r7, r8\n\nexpressions = []\n\nfor i in range(1, len(candidates) + 1):\n    for combo in combinations(candidates, i):\n        for signs in product([-1, 1], repeat=len(combo)):\n            expr = \"df['r6'] == df['r0']\"\n            for sign, row in zip(signs, combo):\n                expr += f\" + {sign} * df['r{row}']\"  # df 中的列名\n            expressions.append(expr)\n\nresults = []\nfor expr in expressions:\n    result = eval(expr)\n    if all(result):\n        print(expr)","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:17:07.944725Z","iopub.execute_input":"2025-01-07T16:17:07.945117Z","iopub.status.idle":"2025-01-07T16:17:08.570198Z","shell.execute_reply.started":"2025-01-07T16:17:07.945081Z","shell.execute_reply":"2025-01-07T16:17:08.568878Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"candidates = [1, 2, 4, 5, 7, 8]  # r1, r2, r4, r5, r7, r8\n\nexpressions = []\n\nfor i in range(1, len(candidates) + 1):\n    for combo in combinations(candidates, i):\n        for signs in product([-1, 1], repeat=len(combo)):\n            expr = \"df['r6'] == df['r3']\"\n            for sign, row in zip(signs, combo):\n                expr += f\" + {sign} * df['r{row}']\"  # df 中的列名\n            expressions.append(expr)\n\nresults = []\nfor expr in expressions:\n    result = eval(expr)\n    if all(result):\n        print(expr)","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:17:08.571668Z","iopub.execute_input":"2025-01-07T16:17:08.572122Z","iopub.status.idle":"2025-01-07T16:17:09.204552Z","shell.execute_reply.started":"2025-01-07T16:17:08.572072Z","shell.execute_reply":"2025-01-07T16:17:09.203230Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features_data = pd.read_csv(r'/kaggle/input/jane-street-real-time-market-data-forecasting/features.csv')\nresponders_data = pd.read_csv(r'/kaggle/input/jane-street-real-time-market-data-forecasting/responders.csv')\nsample_data = pd.read_csv(r'/kaggle/input/jane-street-real-time-market-data-forecasting/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:17:09.207474Z","iopub.execute_input":"2025-01-07T16:17:09.208037Z","iopub.status.idle":"2025-01-07T16:17:09.232201Z","shell.execute_reply.started":"2025-01-07T16:17:09.207859Z","shell.execute_reply":"2025-01-07T16:17:09.230490Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"features_data :\", features_data.shape)\nprint(\"responders_data :\", responders_data.shape)\nprint(\"sample_data :\", sample_data.shape)","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:17:09.234029Z","iopub.execute_input":"2025-01-07T16:17:09.234739Z","iopub.status.idle":"2025-01-07T16:17:09.243151Z","shell.execute_reply.started":"2025-01-07T16:17:09.234687Z","shell.execute_reply":"2025-01-07T16:17:09.241731Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features_data.head()","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:17:09.244763Z","iopub.execute_input":"2025-01-07T16:17:09.245294Z","iopub.status.idle":"2025-01-07T16:17:09.276901Z","shell.execute_reply.started":"2025-01-07T16:17:09.245230Z","shell.execute_reply":"2025-01-07T16:17:09.275664Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features_data.isnull().sum().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:17:09.278304Z","iopub.execute_input":"2025-01-07T16:17:09.278720Z","iopub.status.idle":"2025-01-07T16:17:09.291969Z","shell.execute_reply.started":"2025-01-07T16:17:09.278634Z","shell.execute_reply":"2025-01-07T16:17:09.290626Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"responders_data.head()","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:17:09.293536Z","iopub.execute_input":"2025-01-07T16:17:09.294030Z","iopub.status.idle":"2025-01-07T16:17:09.313537Z","shell.execute_reply.started":"2025-01-07T16:17:09.293993Z","shell.execute_reply":"2025-01-07T16:17:09.312141Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Exclude a specific column, for example 'category2'\ncolumns_to_plot = responders_data.columns.drop('responder')\n\n# Loop through each categorical column and create a pie chart\nfor col in columns_to_plot:\n    plt.figure(figsize=(6, 6))\n    responders_data[col].value_counts().plot.pie(autopct='%1.1f%%', figsize=(5, 5))\n    plt.title(f'Distribution of {col}')  # Add title for each pie chart\n    plt.ylabel('')  # Removes the default y-label\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:17:09.315053Z","iopub.execute_input":"2025-01-07T16:17:09.315524Z","iopub.status.idle":"2025-01-07T16:17:10.092071Z","shell.execute_reply.started":"2025-01-07T16:17:09.315453Z","shell.execute_reply":"2025-01-07T16:17:10.090752Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_data.head()","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:17:10.093715Z","iopub.execute_input":"2025-01-07T16:17:10.094307Z","iopub.status.idle":"2025-01-07T16:17:10.114125Z","shell.execute_reply.started":"2025-01-07T16:17:10.094230Z","shell.execute_reply":"2025-01-07T16:17:10.112725Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import polars as pl","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:23:11.016342Z","iopub.execute_input":"2025-01-07T16:23:11.016765Z","iopub.status.idle":"2025-01-07T16:23:11.104868Z","shell.execute_reply.started":"2025-01-07T16:23:11.016727Z","shell.execute_reply":"2025-01-07T16:23:11.103734Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet\")\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:23:15.560488Z","iopub.execute_input":"2025-01-07T16:23:15.560937Z","iopub.status.idle":"2025-01-07T16:24:07.589289Z","shell.execute_reply.started":"2025-01-07T16:23:15.560896Z","shell.execute_reply":"2025-01-07T16:24:07.587875Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet\")\ntest_data.head()","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:17:57.720322Z","iopub.execute_input":"2025-01-07T16:17:57.720827Z","iopub.status.idle":"2025-01-07T16:17:57.781896Z","shell.execute_reply.started":"2025-01-07T16:17:57.720777Z","shell.execute_reply":"2025-01-07T16:17:57.780384Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import SimpleImputer\nfrom lightgbm import LGBMRegressor\nfrom sklearn.metrics import mean_squared_error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T16:32:27.518306Z","iopub.execute_input":"2025-01-07T16:32:27.518817Z","iopub.status.idle":"2025-01-07T16:32:27.525868Z","shell.execute_reply.started":"2025-01-07T16:32:27.518775Z","shell.execute_reply":"2025-01-07T16:32:27.524580Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX = train_data.drop(['partition_id','responder_0','responder_1','responder_2','responder_3','responder_4','responder_5','responder_6','responder_7','responder_8'])\ny = train_data['responder_6']\ntest_data = test_data.drop(['row_id','is_scored'])","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:32:32.197197Z","iopub.execute_input":"2025-01-07T16:32:32.197674Z","iopub.status.idle":"2025-01-07T16:32:32.230220Z","shell.execute_reply.started":"2025-01-07T16:32:32.197627Z","shell.execute_reply":"2025-01-07T16:32:32.228572Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet\")\ntest_data.head()","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:17:57.720322Z","iopub.execute_input":"2025-01-07T16:17:57.720827Z","iopub.status.idle":"2025-01-07T16:17:57.781896Z","shell.execute_reply.started":"2025-01-07T16:17:57.720777Z","shell.execute_reply":"2025-01-07T16:17:57.780384Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet\")\ntest_data.head()","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:17:57.720322Z","iopub.execute_input":"2025-01-07T16:17:57.720827Z","iopub.status.idle":"2025-01-07T16:17:57.781896Z","shell.execute_reply.started":"2025-01-07T16:17:57.720777Z","shell.execute_reply":"2025-01-07T16:17:57.780384Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.shape, y.shape, test.shape","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:17:57.828027Z","iopub.execute_input":"2025-01-07T16:17:57.828741Z","iopub.status.idle":"2025-01-07T16:17:57.838233Z","shell.execute_reply.started":"2025-01-07T16:17:57.828660Z","shell.execute_reply":"2025-01-07T16:17:57.837052Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(type(X))\nprint(type(X.sample))","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:17:57.840074Z","iopub.execute_input":"2025-01-07T16:17:57.840374Z","iopub.status.idle":"2025-01-07T16:17:57.856985Z","shell.execute_reply.started":"2025-01-07T16:17:57.840343Z","shell.execute_reply":"2025-01-07T16:17:57.855736Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(type(y))","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:17:57.858254Z","iopub.execute_input":"2025-01-07T16:17:57.858605Z","iopub.status.idle":"2025-01-07T16:17:57.872752Z","shell.execute_reply.started":"2025-01-07T16:17:57.858572Z","shell.execute_reply":"2025-01-07T16:17:57.871088Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import polars as pl\n\n# Step 1: Sample 10% of the data in Polars\nX_sample = X.sample(fraction=0.1, seed=42)\n\n# Step 2: Get the row indices of the sampled rows\nsample_indices = X_sample.get_column('date_id').to_numpy().astype(int)\n\n# Step 3: Use these indices to sample y\ny_sample = y[sample_indices]\n\n# Display the results\nprint(X_sample)\nprint(y_sample)","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:17:57.874155Z","iopub.execute_input":"2025-01-07T16:17:57.874745Z","iopub.status.idle":"2025-01-07T16:18:10.045957Z","shell.execute_reply.started":"2025-01-07T16:17:57.874691Z","shell.execute_reply":"2025-01-07T16:18:10.044416Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert Polars DataFrame to Pandas DataFrame for sklearn compatibility\nX_sample_pd = X_sample.to_pandas()\ny_sample_pd = y_sample.to_pandas()","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:22:58.071146Z","iopub.execute_input":"2025-01-07T16:22:58.071629Z","iopub.status.idle":"2025-01-07T16:22:58.094303Z","shell.execute_reply.started":"2025-01-07T16:22:58.071588Z","shell.execute_reply":"2025-01-07T16:22:58.092896Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 4: Define categorical and numerical features\nnumeric_features = X_sample_pd.select_dtypes(include=[np.number]).columns.tolist()\ncategorical_features = X_sample_pd.select_dtypes(exclude=[np.number]).columns.tolist()","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:18:11.020962Z","iopub.execute_input":"2025-01-07T16:18:11.021403Z","iopub.status.idle":"2025-01-07T16:18:13.213751Z","shell.execute_reply.started":"2025-01-07T16:18:11.021354Z","shell.execute_reply":"2025-01-07T16:18:13.212306Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 5: Define preprocessors for numeric and categorical features\nnumeric_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='mean')),  # Impute missing values\n    ('scaler', StandardScaler())                  # Scale numerical features\n])\n\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),  # Impute missing values\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))      # One-hot encode categorical features\n])","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:18:13.214927Z","iopub.execute_input":"2025-01-07T16:18:13.215358Z","iopub.status.idle":"2025-01-07T16:18:13.222464Z","shell.execute_reply.started":"2025-01-07T16:18:13.215312Z","shell.execute_reply":"2025-01-07T16:18:13.221068Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 6: Combine preprocessors into a ColumnTransformer\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numeric_transformer, numeric_features),\n        ('cat', categorical_transformer, categorical_features)\n    ])","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:18:13.224225Z","iopub.execute_input":"2025-01-07T16:18:13.224718Z","iopub.status.idle":"2025-01-07T16:18:13.258007Z","shell.execute_reply.started":"2025-01-07T16:18:13.224665Z","shell.execute_reply":"2025-01-07T16:18:13.256691Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 7: Define the pipeline with preprocessing and LGBMRegressor\npipeline = Pipeline(steps=[\n    ('preprocessor', preprocessor),\n    ('model', LGBMRegressor(random_state=42))  # LightGBM Regressor\n])","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:18:13.259756Z","iopub.execute_input":"2025-01-07T16:18:13.260194Z","iopub.status.idle":"2025-01-07T16:18:13.275363Z","shell.execute_reply.started":"2025-01-07T16:18:13.260147Z","shell.execute_reply":"2025-01-07T16:18:13.273437Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 8: Split data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X_sample_pd, y_sample_pd, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:18:13.277081Z","iopub.execute_input":"2025-01-07T16:18:13.277616Z","iopub.status.idle":"2025-01-07T16:18:20.190825Z","shell.execute_reply.started":"2025-01-07T16:18:13.277567Z","shell.execute_reply":"2025-01-07T16:18:20.189468Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 9: Fit the pipeline\npipeline.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2025-01-07T16:18:20.192216Z","iopub.execute_input":"2025-01-07T16:18:20.192578Z","execution_failed":"2025-01-07T16:18:26.212Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error, r2_score\n# Get predictions\npredictions = pipeline.predict(X_test)\n# Display metrics\nmse = mean_squared_error(y_test, predictions)\nprint(\"MSE:\", mse)\nrmse = np.sqrt(mse)\nprint(\"RMSE:\", rmse)\nr2 = r2_score(y_test, predictions)\nprint(\"R2:\", r2)\n\n# Plot predicted vs actual\nplt.scatter(y_test, predictions)\nplt.xlabel('Actual Labels')\nplt.ylabel('Predicted Labels')\nplt.title('FloodProbability Predictions')\nz = np.polyfit(y_test, predictions, 1)\np = np.poly1d(z)\nplt.plot(y_test,p(y_test), color='magenta')\nplt.show()","metadata":{"trusted":true,"execution":{"execution_failed":"2025-01-07T16:18:26.213Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Assuming you have a Polars DataFrame `test` for making predictions\n# Convert the test DataFrame from Polars to Pandas\nX_test_pd = test.to_pandas()\n\n# Ensure your test DataFrame has the same structure as your training data\n# Use the pipeline to predict\npred = pipeline.predict(X_test_pd)\n\n# Print the predictions\n#print(pred)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T16:22:46.014024Z","iopub.execute_input":"2025-01-07T16:22:46.014437Z","iopub.status.idle":"2025-01-07T16:22:46.041136Z","shell.execute_reply.started":"2025-01-07T16:22:46.014400Z","shell.execute_reply":"2025-01-07T16:22:46.039574Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.DataFrame({'row_id': sample_data.row_id, 'responder_6': pred})\n#print(submission.head())\nsubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T16:22:41.046960Z","iopub.execute_input":"2025-01-07T16:22:41.047427Z","iopub.status.idle":"2025-01-07T16:22:41.358821Z","shell.execute_reply.started":"2025-01-07T16:22:41.047384Z","shell.execute_reply":"2025-01-07T16:22:41.356945Z"}},"outputs":[],"execution_count":null}]}