{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.16","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":84493,"databundleVersionId":11037875,"sourceType":"competition"}],"dockerImageVersionId":30839,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport polars as pl\nimport optuna\nplt.style.use('ggplot')\n\n\nimport gc\nimport xgboost as xgb\n# import lightgbm as lgb\nfrom sklearn.model_selection import train_test_split\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-02T12:49:57.354063Z","iopub.execute_input":"2025-03-02T12:49:57.354452Z","iopub.status.idle":"2025-03-02T12:49:59.983579Z","shell.execute_reply.started":"2025-03-02T12:49:57.354423Z","shell.execute_reply":"2025-03-02T12:49:59.982314Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !pip install xgboost","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T12:49:52.755862Z","iopub.execute_input":"2025-03-02T12:49:52.756266Z","iopub.status.idle":"2025-03-02T12:49:52.760755Z","shell.execute_reply.started":"2025-03-02T12:49:52.756238Z","shell.execute_reply":"2025-03-02T12:49:52.759467Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Reading and looking at the data\n-  The data is quiet large for that I will use polars. If necessary I will also reduce the memory. ","metadata":{}},{"cell_type":"code","source":"    # def reduce_memory_usage(df: pl.DataFrame, name) -> pl.DataFrame:\n    #     print(f\"Memory usage of dataframe \\\"{name}\\\" is {round(df.estimated_size('mb'), 4)} MB,\")\n        \n    #     int_types = [\n    #         pl.Int8,\n    #         pl.Int16,\n    #         pl.Int32,\n    #         pl.Int64,\n    #         pl.UInt8,\n    #         pl.UInt16,\n    #         pl.UInt32,\n    #         pl.UInt64,\n    #     ]\n    #     float_types = [pl.Float32, pl.Float64]\n        \n    #     for col in df.columns:\n    #         col_type = df[col].dtype\n    #         if col_type in int_types:\n    #             c_min = df[col].min()\n    #             c_max = df[col].max()\n                \n    #             if c_min is not None and c_max is not None:\n                    \n    #                 if col_type in int_types:\n    #                     if c_min >= 0:\n    #                         if (c_min >= np.iinfo(np.uint8).min\n    #                             and c_max <= np.iinfo(np.uint8).max):\n                                \n    #                             df = df.with_columns(df[col].cast(pl.UInt8))\n                                \n    #                         elif (c_min >= np.iinfo(np.uint16).min\n    #                             and c_max <= np.iinfo(np.uint16).max):\n                                \n    #                             df = df.with_columns(df[col].cast(pl.UInt16))\n                                \n    #                         elif (\n    #                             c_min >= np.iinfo(np.uint32).min\n    #                             and c_max <= np.iinfo(np.uint32).max\n    #                         ):\n    #                             df = df.with_columns(df[col].cast(pl.UInt32))\n    #                         elif (\n    #                             c_min >= np.iinfo(np.uint64).min\n    #                             and c_max <= np.iinfo(np.uint64).max\n    #                         ):\n    #                             df = df.with_columns(df[col].cast(pl.UInt64))\n    #                     else:\n    #                         if (\n    #                             c_min >= np.iinfo(np.int8).min\n    #                             and c_max <= np.iinfo(np.int8).max\n    #                         ):\n    #                             df = df.with_columns(df[col].cast(pl.Int8))\n    #                         elif (\n    #                             c_min >= np.iinfo(np.int16).min\n    #                             and c_max <= np.iinfo(np.int16).max\n    #                         ):\n    #                             df = df.with_columns(df[col].cast(pl.Int16))\n    #                         elif (\n    #                             c_min >= np.iinfo(np.int32).min\n    #                             and c_max <= np.iinfo(np.int32).max\n    #                         ):\n    #                             df = df.with_columns(df[col].cast(pl.Int32))\n    #                         elif (\n    #                             c_min >= np.iinfo(np.int64).min\n    #                             and c_max <= np.iinfo(np.int64).max\n    #                         ):\n    #                             df = df.with_columns(df[col].cast(pl.Int64))\n    #                 elif col_type in float_types:\n    #                     if (\n    #                         c_min > np.finfo(np.float32).min\n    #                         and c_max < np.finfo(np.float32).max\n    #                     ):\n    #                         df = df.with_columns(df[col].cast(pl.Float32))\n                                \n                                \n    #     print(\n    #         f\"Memory usage of dataframe \\\"{name}\\\" became {round(df.estimated_size('mb'), 4)} MB.\"\n    #     )\n\n    #     return df\n    \n    \n    # def to_pandas(df: pl.DataFrame, cat_cols: list[str] = None) -> (pd.DataFrame, list[str]):\n    #     df: pd.DataFrame = df.to_pandas()\n\n    #     if cat_cols is None:\n    #         cat_cols = list(df.select_dtypes(\"object\").columns)\n\n    #     df[cat_cols] = df[cat_cols].astype(\"str\")                       \n\n    #     return df, cat_cols\n\n\n\n\n# def reduce_memory_usage() -> pl.Expr:\n#     expressions = [\n#         pl.col(pl.Float64).cast(pl.Float32),\n#         pl.col(\"date_id\", \"time_id\").cast(pl.Int16),\n#         pl.col(\"symbol_id\").cast(pl.Int8),\n#         ]\n#     return expressions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T09:31:20.951684Z","iopub.status.idle":"2025-03-02T09:31:20.952066Z","shell.execute_reply.started":"2025-03-02T09:31:20.951842Z","shell.execute_reply":"2025-03-02T09:31:20.951882Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_pl = []\n\nfor i in range(9):\n    df_pl = pl.read_parquet(f'/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={i}/part-0.parquet')\n    sample_pl.append(df_pl)\n\n\ndf_pl = pl.concat(sample_pl)\ndf_pl.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T12:50:02.491621Z","iopub.execute_input":"2025-03-02T12:50:02.492353Z","iopub.status.idle":"2025-03-02T12:50:16.084440Z","shell.execute_reply.started":"2025-03-02T12:50:02.492318Z","shell.execute_reply":"2025-03-02T12:50:16.083247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_pl.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T10:21:05.781252Z","iopub.execute_input":"2025-03-02T10:21:05.781558Z","iopub.status.idle":"2025-03-02T10:21:05.786482Z","shell.execute_reply.started":"2025-03-02T10:21:05.781533Z","shell.execute_reply":"2025-03-02T10:21:05.785380Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_pl.glimpse()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-15T10:28:24.064670Z","iopub.execute_input":"2025-02-15T10:28:24.064979Z","iopub.status.idle":"2025-02-15T10:28:24.073241Z","shell.execute_reply.started":"2025-02-15T10:28:24.064951Z","shell.execute_reply":"2025-02-15T10:28:24.072318Z"},"scrolled":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## looking at the features with missing values. ","metadata":{}},{"cell_type":"code","source":"null_counts = df_pl.null_count().transpose(include_header=True)\nnull_counts = null_counts.to_pandas()\nnull_counts.rename(columns={'column':'Feature', 'column_0':'Count'}, inplace=True)\nnull_counts = null_counts[null_counts['Count']>0]\nnull_counts = null_counts.sort_values('Count',ascending=False)\nnull_counts['Percentage'] = round((null_counts['Count']/40852762)*100,2)\nnull_counts","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T16:31:45.332421Z","iopub.execute_input":"2025-02-05T16:31:45.332630Z","iopub.status.idle":"2025-02-05T16:31:45.425225Z","shell.execute_reply.started":"2025-02-05T16:31:45.332600Z","shell.execute_reply":"2025-02-05T16:31:45.424293Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(70, 10))\nplt.bar(null_counts['Feature'], null_counts['Count'])\nplt.ylabel('count (million)', fontsize=25)\nplt.xlabel('Features', fontsize=25)\nplt.suptitle('Missing value counts', fontsize=40)\nplt.yticks(fontsize=25)\nplt.xticks(fontsize=13)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T16:31:45.427013Z","iopub.execute_input":"2025-02-05T16:31:45.427307Z","iopub.status.idle":"2025-02-05T16:31:46.256020Z","shell.execute_reply.started":"2025-02-05T16:31:45.427284Z","shell.execute_reply":"2025-02-05T16:31:46.255105Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"An analysis of the dataset reveals that out of 92 columns, 47 contain missing values. Notably, four features—feature_26, feature_27, feature_31, and feature_21—have approximately **20% of their data missing**.\n\nAdditionally, three features—feature_42, feature_50, and feature_53—exhibit around 9.5% missing data, while five features—feature_03, feature_01, feature_02, feature_04, and feature_00—have approximately **7% missing data**.\n\nThe remaining features within the set of 47 have lower percentages of missing values, ranging from **2% to less than 1%**.","metadata":{"execution":{"iopub.status.busy":"2025-02-05T02:35:10.119895Z","iopub.execute_input":"2025-02-05T02:35:10.120330Z","iopub.status.idle":"2025-02-05T02:35:10.128345Z","shell.execute_reply.started":"2025-02-05T02:35:10.120295Z","shell.execute_reply":"2025-02-05T02:35:10.126744Z"}}},{"cell_type":"code","source":"# df_pl_sample = df_pl[:10000000]\n# df_pl_sample_symbl14 = df_pl_sample.filter(pl.col('symbol_id')==14)\n# df_pl_sample_symbl14","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T02:35:20.050624Z","iopub.execute_input":"2025-02-05T02:35:20.050973Z","iopub.status.idle":"2025-02-05T02:35:20.055051Z","shell.execute_reply.started":"2025-02-05T02:35:20.050943Z","shell.execute_reply":"2025-02-05T02:35:20.053805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# sns.barplot(x=df_pl.group_by('symbol_id').count()['symbol_id'],y=df_pl.group_by('symbol_id').count()['count'])\n# kk = df_pl.group_by('symbol_id').count().to_pandas()\n# sns.lineplot(x='symbol_id',y='count', data=kk)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T02:35:54.854207Z","iopub.execute_input":"2025-02-05T02:35:54.854614Z","iopub.status.idle":"2025-02-05T02:35:54.858734Z","shell.execute_reply.started":"2025-02-05T02:35:54.854580Z","shell.execute_reply":"2025-02-05T02:35:54.857514Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Type of Visualizations we can make here\n1. Understanding data distribution - Distribution of Features **📌 Goal: See how market features (feature_0 to feature_78) and responder_6 are distributed.**\n2. Time-Series Analysis - Trends in Responder_6 Over Time **📌 Goal: Understand how responder_6 behaves over different time periods.**\n3.  Feature Correlation - Heatmap of Correlations **📌 Goal: Identify correlated features to reduce redundancy.**\n5.  Symbol-Based Analysis - Activity of Different Stocks **📌 Goal: See how active different symbol_id stocks are.**\n6.  Weighting Effect - Distribution of Weights **📌 Goal: Check if certain rows (stocks or time periods) are overweighted in scoring.**\n7.  Feature-Target Relationship - Boxplot of Features vs. Responder_6 **📌 Goal: See how responder values change across different feature ranges.**\n8.   Dimensionality Reduction - PCA Visualization **📌 Goal: Check if features can be reduced in dimension.**","metadata":{}},{"cell_type":"markdown","source":"### 1.Understanding data distribution\n*plot distribution of few features*","metadata":{}},{"cell_type":"code","source":"# Plot distribution of a few features\nfeatures_to_plot = ['feature_00', 'feature_10', 'feature_20']\nfor feature in features_to_plot:\n    plt.figure(figsize=(8, 4))\n    sns.histplot(df_pl[feature], bins=50, kde=True)\n    plt.title(f\"Distribution of {feature}\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T02:55:57.930951Z","iopub.execute_input":"2025-02-05T02:55:57.931356Z","iopub.status.idle":"2025-02-05T03:04:59.500916Z","shell.execute_reply.started":"2025-02-05T02:55:57.931324Z","shell.execute_reply":"2025-02-05T03:04:59.499826Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features_to_plot = ['feature_04', 'feature_08', 'feature_17','feature_21','feature_32','feature_46','feature_66']\nfor feature in features_to_plot:\n    plt.figure(figsize=(8, 4))\n    sns.histplot(df_pl[feature], bins=50, kde=True)\n    plt.title(f\"Distribution of {feature}\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T03:14:34.094951Z","iopub.execute_input":"2025-02-05T03:14:34.095419Z","iopub.status.idle":"2025-02-05T03:35:44.057857Z","shell.execute_reply.started":"2025-02-05T03:14:34.095381Z","shell.execute_reply":"2025-02-05T03:35:44.056909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features_to_plot = ['responder_2', 'responder_6', 'responder_8']\nfor feature in features_to_plot:\n    plt.figure(figsize=(8, 4))\n    sns.histplot(df_pl[feature], bins=50, kde=True)\n    plt.title(f\"Distribution of {feature}\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T03:39:03.682784Z","iopub.execute_input":"2025-02-05T03:39:03.683306Z","iopub.status.idle":"2025-02-05T03:47:56.644472Z","shell.execute_reply.started":"2025-02-05T03:39:03.683264Z","shell.execute_reply":"2025-02-05T03:47:56.643258Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 2.Time-Series Analysis\n-  Trends in Responder_6 Over Time","metadata":{}},{"cell_type":"code","source":"date_responder6 = df_pl.select(['date_id','time_id', 'responder_6']).to_pandas()\ndate_responder6","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-06T08:57:45.842018Z","iopub.execute_input":"2025-02-06T08:57:45.842335Z","iopub.status.idle":"2025-02-06T08:57:46.506575Z","shell.execute_reply.started":"2025-02-06T08:57:45.842308Z","shell.execute_reply":"2025-02-06T08:57:46.505709Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# date_responder6['date'] = pd.to_datetime(date_responder6['date'])\n# date_responder6.set_index('date_id', inplace=True)\n\nplt.figure(figsize=(16, 5))\ndate_responder6['responder_6'].rolling(window=1000).mean().plot(color='blue',linewidth =0.05)\nplt.title(\"Rolling Mean of Responder_6 Over Time\")\nplt.xlabel(\"Date\")\nplt.ylabel(\"Responder_6\")\n# plt.grid(color = 'lightgrey' , linewidth=0.8)\nplt.axhline(0, color='green', linestyle='-', linewidth=1.2)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-06T08:57:50.140606Z","iopub.execute_input":"2025-02-06T08:57:50.140931Z","iopub.status.idle":"2025-02-06T08:58:00.254191Z","shell.execute_reply.started":"2025-02-06T08:57:50.140904Z","shell.execute_reply":"2025-02-06T08:58:00.253025Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Feature Correlation\n- Heatmap of correlations","metadata":{}},{"cell_type":"code","source":"df_pl.corr()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-15T10:29:00.084888Z","iopub.execute_input":"2025-02-15T10:29:00.085182Z","execution_failed":"2025-02-15T10:29:12.775Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_pl_small_corr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-11T15:28:43.359253Z","iopub.execute_input":"2025-02-11T15:28:43.359643Z","iopub.status.idle":"2025-02-11T15:28:43.387836Z","shell.execute_reply.started":"2025-02-11T15:28:43.359617Z","shell.execute_reply":"2025-02-11T15:28:43.386681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_pl_small_corr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-15T11:31:16.529050Z","iopub.execute_input":"2025-02-15T11:31:16.529358Z","iopub.status.idle":"2025-02-15T11:31:16.570670Z","shell.execute_reply.started":"2025-02-15T11:31:16.529333Z","shell.execute_reply":"2025-02-15T11:31:16.569694Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# trying to get a rough idea of the features correlation by using a small subset of data\n# df_pl_small = df_pl[:9000000].to_pandas()\n# df_pl_small_corr = df_pl_small.corr()\n# df_pl_small_corr.drop(['date_id', 'time_id', 'symbol_id'], axis=1, inplace=True)\n# df_pl_small_corr.drop(['date_id', 'time_id', 'symbol_id'], axis=0, inplace=True)\n\n\n# Plot heatmap\nplt.figure(figsize=(96, 48))\nsns.heatmap(df_pl_small_corr, cmap=\"coolwarm\", center=0, vmin=-1, vmax=1, annot=True, linewidth=0.05)\nplt.title(\"All Feature Correlation Heatmap\")\nplt.xticks(fontsize=20)\nplt.yticks(fontsize=20)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T06:18:39.267362Z","iopub.execute_input":"2025-02-16T06:18:39.267658Z","iopub.status.idle":"2025-02-16T06:18:55.955279Z","shell.execute_reply.started":"2025-02-16T06:18:39.267633Z","shell.execute_reply":"2025-02-16T06:18:55.953785Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Corealation between Features\n- Feature_73 - Feature_74 (.96)\n- Feature_77 - Feature_78 (.96)\n- Feature_67 - Feature_12 (.93)\n- Feature_70 - Feature_12 (.91)\n- Feature_68 - Feature_13 (.81)\n- Feature_72 - Feature_14 (.86)\n- Feature_69 - Feature_14 (.89)\n- Feature_15 - Feature_30 (.89)\n- Feature_15 - Feature_17 (.92)\n- Feature_15 - Feature_29 (.8)\n- Feature_31 - Feature_21 (.97)\n- Feature_00 - Feature_02 (.92)\n- Feature_22 - weight (.88)\n- Feature_40 - Feature_43 (.89)\n- Feature_40 - Feature_41 (.81)\n- Feature_35 - Feature_34 (.91)\n- Feature_35 - Feature_32 (.86)\n- Feature_34 - Feature_32 (.84)\nTo ber continued...","metadata":{}},{"cell_type":"code","source":"#After identifying correlated features from a sample subset of the dataset, a heatmap is applied to assess whether the correlation holds.\n\ncorr_matrix = df_pl.select(['feature_12','feature_67','feature_70','feature_68', \\\n                           'feature_73','feature_74','feature_77','feature_78','feature_69',]).to_pandas().corr()\n\n# Plot heatmap\nplt.figure(figsize=(10, 6))\nsns.heatmap(corr_matrix, cmap=\"crest\", center=0, vmin=-1, vmax=1,annot=True, linewidth=0.5)\nplt.title(\"Feature Correlation Heatmap\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T06:25:23.457869Z","iopub.execute_input":"2025-02-16T06:25:23.458247Z","iopub.status.idle":"2025-02-16T06:25:36.901643Z","shell.execute_reply.started":"2025-02-16T06:25:23.458218Z","shell.execute_reply":"2025-02-16T06:25:36.900818Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#After identifying correlated features from a sample subset of the dataset, a heatmap is applied to assess whether the correlation holds.\n\ncorr_matrix = df_pl.select(['feature_00','feature_02','feature_21','feature_31', \\\n                           'feature_32','feature_34','feature_35','feature_15','feature_17','feature_29']).to_pandas().corr()\n\n# Plot heatmap\nplt.figure(figsize=(10, 6))\nsns.heatmap(corr_matrix, cmap=\"coolwarm\",  center=0, vmin=-1, vmax=1,annot=True, linewidth=0.5)\nplt.title(\"Feature Correlation Heatmap\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T06:32:15.369332Z","iopub.execute_input":"2025-02-16T06:32:15.369622Z","iopub.status.idle":"2025-02-16T06:32:15.834244Z","shell.execute_reply.started":"2025-02-16T06:32:15.369600Z","shell.execute_reply":"2025-02-16T06:32:15.833391Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### We will be doing more EDA & Visualizations on the features but before that let's create and tune a good model for this data.","metadata":{}},{"cell_type":"code","source":"# df_pl.filter((pl.col('date_id')>500) & (pl.col('date_id')<=700))\n# df_pl.filter(pl.col('date_id')>1500)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T12:28:59.187724Z","iopub.execute_input":"2025-02-16T12:28:59.188312Z","iopub.status.idle":"2025-02-16T12:28:59.191552Z","shell.execute_reply.started":"2025-02-16T12:28:59.188275Z","shell.execute_reply":"2025-02-16T12:28:59.190668Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Removing Highly correlated Features\n\"\"\"Feature_73 - Feature_74 (.96)\n- Feature_77 - Feature_78 (.96)\n- Feature_67 - Feature_12 (.93)\n- Feature_70 - Feature_12 (.91)\n- Feature_68 - Feature_13 (.81)\n- Feature_72 - Feature_14 (.86)\n- Feature_69 - Feature_14 (.89)\n- Feature_15 - Feature_30 (.89)\n- Feature_15 - Feature_17 (.92)\n- Feature_15 - Feature_29 (.8)\n- Feature_31 - Feature_21 (.97)\n- Feature_00 - Feature_02 (.92)\n- Feature_22 - weight (.88)\n- Feature_40 - Feature_43 (.89)\n- Feature_40 - Feature_41 (.81)\n- Feature_35 - Feature_34 (.91)\n- Feature_35 - Feature_32 (.86)\n- Feature_34 - Feature_32 (.84)\"\"\"\n\n# Will add more features to it after the first model run\n\nFeatures_to_remove = ['feature_74','feature_78','feature_12','feature_70','feature_13','feature_14','feature_69','feature_30',\\\n                     'feature_17','feature_29','feature_21','feature_02','feature_43',\\\n                      'feature_41','feature_34','feature_32']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T09:21:55.541267Z","iopub.execute_input":"2025-03-02T09:21:55.541600Z","iopub.status.idle":"2025-03-02T09:21:55.546415Z","shell.execute_reply.started":"2025-03-02T09:21:55.541573Z","shell.execute_reply":"2025-03-02T09:21:55.545559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" df_pl.filter(pl.col('date_id')>1520)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T11:06:48.040802Z","iopub.execute_input":"2025-03-02T11:06:48.041229Z","iopub.status.idle":"2025-03-02T11:06:48.993967Z","shell.execute_reply.started":"2025-03-02T11:06:48.041196Z","shell.execute_reply":"2025-03-02T11:06:48.992919Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"Keeping some data for test untouched. We have 1529 days of stock data. \nOut of which I am keeping 1400 days for train and validation and 129 days keeping untouched for test data.\"\"\"\n\nTest_data = df_pl.filter(pl.col('date_id')>1520)\nTest_data = Test_data.sample(n=1000, seed = 0)\n\nTrain_data = df_pl.filter(pl.col('date_id')<=1520)\n\n# spliting the train data into train and validation data.\nfeature_names = [f\"feature_{i:02d}\" for i in range(79)]\n# new_features = [x for x in feature_names if x not in Features_to_remove]\n\n\nX_train = Train_data.filter(pl.col('date_id')<=1200).select(feature_names)\nY_train = Train_data.filter(pl.col('date_id')<=1200).select(['responder_6'])\nW_train = Train_data.filter(pl.col('date_id')<=1200).select(['weight'])\n\n\nX_val = Train_data.filter((pl.col('date_id')>1200) & (pl.col('date_id')<=1520)).select(feature_names)\nY_val = Train_data.filter((pl.col('date_id')>1200) & (pl.col('date_id')<=1520)).select(['responder_6'])\nW_val = Train_data.filter((pl.col('date_id')>1200) & (pl.col('date_id')<=1520)).select(['weight'])\n\n\nX_test = Test_data.select(feature_names)\nY_test = Test_data.select(['responder_6'])\nW_test = Test_data.select(['weight'])\n\n\n# Deleteing the main data and Train data to save memory\ndel  Train_data, Test_data\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T12:50:40.368361Z","iopub.execute_input":"2025-03-02T12:50:40.368777Z","iopub.status.idle":"2025-03-02T12:50:40.663134Z","shell.execute_reply.started":"2025-03-02T12:50:40.368750Z","shell.execute_reply":"2025-03-02T12:50:40.661847Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Custom R2 metric for XGBoost\ndef r2_xgb(y_true, y_pred, sample_weight):    \n    r2 = 1 - np.average((y_pred - y_true) ** 2, weights=sample_weight) / (np.average((y_true) ** 2, weights=sample_weight) + 1e-38)\n    return -r2\n\n\nmodel = xgb.XGBRegressor(n_estimators=1600, \\\n                 learning_rate=0.05,\\\n                 max_depth=12, \\\n                 # tree_method='gpu_hist',\\\n                 # device=\"cuda\",\\\n                 objective='reg:squarederror',\\\n                 eval_metric=r2_xgb,\\\n                 verbosity=3, \\\n                 # disable_default_eval_metric=True, \n          early_stopping_rounds=50)\n\n\n# Train XGBoost model with early stopping and verbose logging\nmodel.fit(X_train, Y_train, sample_weight=W_train, \n          eval_set=[(X_val, Y_val)], \n          sample_weight_eval_set=[W_val], \n          verbose=True)","metadata":{"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error\n\n# Define the Optuna objective function\ndef objective(trial):\n    params = {\n        \"objective\": \"reg:squarederror\",\n        \"eval_metric\": 'rmse',\n        # \"tree_method\": \"gpu_hist\",  # Use GPU if available\n        \"lambda\": trial.suggest_loguniform(\"lambda\", 1e-3, 10.0),\n        \"alpha\": trial.suggest_loguniform(\"alpha\", 1e-3, 10.0),\n        \"colsample_bytree\": trial.suggest_uniform(\"colsample_bytree\", 0.4, 1.0),\n        \"subsample\": trial.suggest_uniform(\"subsample\", 0.4, 1.0),\n        \"learning_rate\": trial.suggest_loguniform(\"learning_rate\", 0.01, 0.3),\n        \"n_estimators\": trial.suggest_int(\"n_estimators\", 100, 2000),\n        \"max_depth\": trial.suggest_int(\"max_depth\", 3, 15),\n        \"min_child_weight\": trial.suggest_loguniform(\"min_child_weight\", 1, 10),\n        \"gamma\": trial.suggest_loguniform(\"gamma\", 1e-3, 5.0),\n    }\n\n    # Train model\n    model = xgb.XGBRegressor(**params, early_stopping_rounds=50)\n    model.fit(\n        X_train, Y_train,\n        sample_weight=W_train,  # Include weight\n        eval_set=[(X_val, Y_val)],\n        sample_weight_eval_set=[W_val],\n        verbose=True\n    )\n\n    # Predict & Evaluate\n    preds = model.predict(X_val)\n    rmse = mean_squared_error(Y_val, preds, squared=False)\n    r2_xgb = r2_xgb(Y_val, preds,W_val)\n    return rmse, r2_xgb\n\n# Run Optuna\nstudy = optuna.create_study(direction=\"minimize\")\nstudy.optimize(objective, n_trials=50)  # Increase for better tuning\n\n# pruning_callback = optuna.integration.XGBoostPruningCallback(trial, \"validation-auc\")\n\n# Best parameters\nbest_params = study.best_params\nprint(\"Best parameters:\", best_params)\n\n# Train final model with best parameters\nbest_model = xgb.XGBRegressor(**best_params)\nbest_model.fit(X_train, y_train, sample_weight=W_train)\n\n# Save model\nbest_model.save_model(\"xgb_best_model.json\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T12:50:54.663901Z","iopub.execute_input":"2025-03-02T12:50:54.664335Z","iopub.status.idle":"2025-03-02T13:10:56.216878Z","shell.execute_reply.started":"2025-03-02T12:50:54.664292Z","shell.execute_reply":"2025-03-02T13:10:56.215630Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# model.get_xgb_params()\nfeature_imp = pd.DataFrame({'Feature':feature_names,\n    'value':model.feature_importances_})\nfeature_imp.sort_values(by=['value'], ascending =False, inplace=True)\n\n\nplt.figure(figsize=(16, 6))\nsns.barplot(x=feature_imp['Feature'], y=feature_imp['value'], alpha=0.8)\nplt.xlabel(\"Features\")\nplt.ylabel(\"data\")\nplt.xticks(fontsize=8, rotation=90)\nplt.title(\"Feature Importances\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T11:24:59.968454Z","iopub.execute_input":"2025-03-02T11:24:59.968829Z","iopub.status.idle":"2025-03-02T11:25:00.930388Z","shell.execute_reply.started":"2025-03-02T11:24:59.968800Z","shell.execute_reply":"2025-03-02T11:25:00.928636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make predictions\nY_pred = model.predict(X_test)\nY_pred = pl.from_numpy(Y_pred, schema=[\"Pred\"])\n\n\n# Evaluate model\nr2_xgb = r2_xgb(Y_test, Y_pred,W_test)\nprint(f\"Baseline Xgboost R2 score: {r2_xgb}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T11:40:42.475806Z","iopub.execute_input":"2025-03-02T11:40:42.476120Z","iopub.status.idle":"2025-03-02T11:40:42.520409Z","shell.execute_reply.started":"2025-03-02T11:40:42.476094Z","shell.execute_reply":"2025-03-02T11:40:42.518561Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Y_pred.to_series().shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T12:01:04.339397Z","iopub.execute_input":"2025-03-02T12:01:04.339819Z","iopub.status.idle":"2025-03-02T12:01:04.344837Z","shell.execute_reply.started":"2025-03-02T12:01:04.339786Z","shell.execute_reply":"2025-03-02T12:01:04.343861Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot predicted vs. actual\nplt.figure(figsize=(8, 6))\nsns.scatterplot(x=Y_test.to_series(), y=Y_pred.to_series(), alpha=0.5)\nplt.xlabel(\"Actual Responder_6\")\nplt.ylabel(\"Predicted Responder_6\")\nplt.title(\"Actual vs Predicted Responder_6\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T12:01:37.525431Z","iopub.execute_input":"2025-03-02T12:01:37.525830Z","iopub.status.idle":"2025-03-02T12:01:37.709580Z","shell.execute_reply.started":"2025-03-02T12:01:37.525803Z","shell.execute_reply":"2025-03-02T12:01:37.708813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Create dataset format for LightGBM\ntrain_data = lgb.Dataset(X_train, label=Y_train,weight = W_train)\ntest_data = lgb.Dataset(X_val, label=Y_val,weight = W_val)\n\n# Custom R2 metric for LightGBM\ndef r2_lgb(y_true, y_pred, sample_weight):\n    r2 = 1 - np.average((y_pred - y_true) ** 2, weights=sample_weight) / (np.average((y_true) ** 2, weights=sample_weight) + 1e-38)\n    return 'r2', r2, True\n\n\n# Define parameters (basic tuning)\nparams = {\n    'objective': 'regression',\n    'metric': 'rmse',\n    'boosting_type': 'gbdt',\n    'learning_rate': 0.05,\n    'num_leaves': 256,\n    'max_depth': -1,\n    'device': 'gpu',  # Enable GPU\n    'gpu_platform_id': 0,  # Adjust if multiple GPUs\n    'gpu_device_id': 0\n}\n\n# Train model\nmodel = lgb.train(params, train_data, valid_sets=[test_data], num_boost_round=1000)\n\n# model_lgb = lgb.LGBMRegressor(n_estimators=500, device='gpu', gpu_use_dp=True, objective='l2')\n\n# model_lgb.fit(train_data ,eval_set=[test_data], \n#           callbacks=[\n#               lgb.early_stopping(100), \n#               lgb.log_evaluation(10)\n#           ])\n\n# model_lgb.fit(X_train, Y_train, W_train.to_numpy(),  \n#           eval_metric=[r2_lgb],\n#           eval_set=[(X_val, Y_val, W_val.to_numpy())], \n#           callbacks=[\n#               lgb.early_stopping(100), \n#               lgb.log_evaluation(10)\n#           ])\n\n\n# # Make predictions\n# # y_pred = model.predict(X_test)\n\n# # Evaluate model\n# rmse = mean_squared_error(y_test, y_pred, squared=False)\n# print(f\"Baseline LightGBM RMSE: {rmse}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T10:53:43.930051Z","iopub.execute_input":"2025-02-16T10:53:43.930403Z","execution_failed":"2025-02-16T10:55:04.006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}