{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-17T16:57:17.268280Z","iopub.execute_input":"2024-10-17T16:57:17.268847Z","iopub.status.idle":"2024-10-17T16:57:17.740582Z","shell.execute_reply.started":"2024-10-17T16:57:17.268790Z","shell.execute_reply":"2024-10-17T16:57:17.739181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import SimpleImputer\nfrom lightgbm import LGBMRegressor\nfrom sklearn.metrics import mean_squared_error","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:57:17.742509Z","iopub.execute_input":"2024-10-17T16:57:17.742998Z","iopub.status.idle":"2024-10-17T16:57:19.604853Z","shell.execute_reply.started":"2024-10-17T16:57:17.742959Z","shell.execute_reply":"2024-10-17T16:57:19.603663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_data = pd.read_csv(r'/kaggle/input/jane-street-real-time-market-data-forecasting/features.csv')\nresponders_data = pd.read_csv(r'/kaggle/input/jane-street-real-time-market-data-forecasting/responders.csv')\nsample_data = pd.read_csv(r'/kaggle/input/jane-street-real-time-market-data-forecasting/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:57:19.606557Z","iopub.execute_input":"2024-10-17T16:57:19.607316Z","iopub.status.idle":"2024-10-17T16:57:19.623319Z","shell.execute_reply.started":"2024-10-17T16:57:19.607262Z","shell.execute_reply":"2024-10-17T16:57:19.622218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"features_data :\", features_data.shape)\nprint(\"responders_data :\", responders_data.shape)\nprint(\"sample_data :\", sample_data.shape)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:57:19.626963Z","iopub.execute_input":"2024-10-17T16:57:19.627471Z","iopub.status.idle":"2024-10-17T16:57:19.634478Z","shell.execute_reply.started":"2024-10-17T16:57:19.627414Z","shell.execute_reply":"2024-10-17T16:57:19.632750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:57:19.636659Z","iopub.execute_input":"2024-10-17T16:57:19.637301Z","iopub.status.idle":"2024-10-17T16:57:19.665637Z","shell.execute_reply.started":"2024-10-17T16:57:19.637245Z","shell.execute_reply":"2024-10-17T16:57:19.664338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_data.isnull().sum().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:57:19.667570Z","iopub.execute_input":"2024-10-17T16:57:19.668050Z","iopub.status.idle":"2024-10-17T16:57:19.678976Z","shell.execute_reply.started":"2024-10-17T16:57:19.668000Z","shell.execute_reply":"2024-10-17T16:57:19.677512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Exclude a specific column, for example 'category2'\ncolumns_to_plot = features_data.columns.drop('feature')\n\n# Loop through each categorical column and create a pie chart\nfor col in columns_to_plot:\n    plt.figure(figsize=(6, 6))\n    features_data[col].value_counts().plot.pie(autopct='%1.1f%%', figsize=(5, 5))\n    plt.title(f'Distribution of {col}')  # Add title for each pie chart\n    plt.ylabel('')  # Removes the default y-label\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:57:19.681316Z","iopub.execute_input":"2024-10-17T16:57:19.681908Z","iopub.status.idle":"2024-10-17T16:57:22.346730Z","shell.execute_reply.started":"2024-10-17T16:57:19.681851Z","shell.execute_reply":"2024-10-17T16:57:22.345049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  object datatype columns encoding:\nfrom sklearn.preprocessing import LabelEncoder\nlabelencoder = LabelEncoder()\nfor col_name in features_data.columns:\n    features_data[col_name]=labelencoder.fit_transform(features_data[col_name]).astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:57:22.349149Z","iopub.execute_input":"2024-10-17T16:57:22.349708Z","iopub.status.idle":"2024-10-17T16:57:22.371763Z","shell.execute_reply.started":"2024-10-17T16:57:22.349652Z","shell.execute_reply":"2024-10-17T16:57:22.370231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the correlation matrix\ncorr = features_data.corr()\n\n# Create the heatmap with seaborn\nplt.figure(figsize=(10, 8))\nsns.heatmap(corr, annot=True, cmap='coolwarm', fmt=\".2f\")\n\n# Display the heatmap\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:57:22.373628Z","iopub.execute_input":"2024-10-17T16:57:22.374478Z","iopub.status.idle":"2024-10-17T16:57:23.475874Z","shell.execute_reply.started":"2024-10-17T16:57:22.374422Z","shell.execute_reply":"2024-10-17T16:57:23.474508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"responders_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:57:23.480917Z","iopub.execute_input":"2024-10-17T16:57:23.481481Z","iopub.status.idle":"2024-10-17T16:57:23.497185Z","shell.execute_reply.started":"2024-10-17T16:57:23.481420Z","shell.execute_reply":"2024-10-17T16:57:23.495888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"responders_data.isnull().sum().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:57:23.498549Z","iopub.execute_input":"2024-10-17T16:57:23.498943Z","iopub.status.idle":"2024-10-17T16:57:23.511417Z","shell.execute_reply.started":"2024-10-17T16:57:23.498897Z","shell.execute_reply":"2024-10-17T16:57:23.510234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Exclude a specific column, for example 'category2'\ncolumns_to_plot = responders_data.columns.drop('responder')\n\n# Loop through each categorical column and create a pie chart\nfor col in columns_to_plot:\n    plt.figure(figsize=(6, 6))\n    responders_data[col].value_counts().plot.pie(autopct='%1.1f%%', figsize=(5, 5))\n    plt.title(f'Distribution of {col}')  # Add title for each pie chart\n    plt.ylabel('')  # Removes the default y-label\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:57:23.512907Z","iopub.execute_input":"2024-10-17T16:57:23.513305Z","iopub.status.idle":"2024-10-17T16:57:24.032331Z","shell.execute_reply.started":"2024-10-17T16:57:23.513265Z","shell.execute_reply":"2024-10-17T16:57:24.031120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  object datatype columns encoding:\nfrom sklearn.preprocessing import LabelEncoder\nlabelencoder = LabelEncoder()\nfor col_name in responders_data.columns:\n    responders_data[col_name]=labelencoder.fit_transform(responders_data[col_name]).astype(int)\n    ","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:57:24.034309Z","iopub.execute_input":"2024-10-17T16:57:24.034883Z","iopub.status.idle":"2024-10-17T16:57:24.048079Z","shell.execute_reply.started":"2024-10-17T16:57:24.034830Z","shell.execute_reply":"2024-10-17T16:57:24.046547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the correlation matrix\ncorr = responders_data.corr()\n\n# Create the heatmap with seaborn\nplt.figure(figsize=(10, 8))\nsns.heatmap(corr, annot=True, cmap='coolwarm', fmt=\".2f\")\n\n# Display the heatmap\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:57:24.050976Z","iopub.execute_input":"2024-10-17T16:57:24.052446Z","iopub.status.idle":"2024-10-17T16:57:24.421731Z","shell.execute_reply.started":"2024-10-17T16:57:24.052369Z","shell.execute_reply":"2024-10-17T16:57:24.420472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:57:24.423356Z","iopub.execute_input":"2024-10-17T16:57:24.423840Z","iopub.status.idle":"2024-10-17T16:57:24.435448Z","shell.execute_reply.started":"2024-10-17T16:57:24.423792Z","shell.execute_reply":"2024-10-17T16:57:24.434251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import polars as pl","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:57:24.437001Z","iopub.execute_input":"2024-10-17T16:57:24.437373Z","iopub.status.idle":"2024-10-17T16:57:24.819506Z","shell.execute_reply.started":"2024-10-17T16:57:24.437334Z","shell.execute_reply":"2024-10-17T16:57:24.818174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet\")\ntrain_data = train_data.sample(fraction=0.1, seed=42)\ntrain_data = train_data.to_pandas()\nprint(train_data.shape)\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:57:24.820921Z","iopub.execute_input":"2024-10-17T16:57:24.821261Z","iopub.status.idle":"2024-10-17T16:58:48.936483Z","shell.execute_reply.started":"2024-10-17T16:57:24.821227Z","shell.execute_reply":"2024-10-17T16:58:48.935263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet\")\ntest_data = test_data.to_pandas()\nprint(test_data.shape)\ntest_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:58:48.938013Z","iopub.execute_input":"2024-10-17T16:58:48.938393Z","iopub.status.idle":"2024-10-17T16:58:48.980265Z","shell.execute_reply.started":"2024-10-17T16:58:48.938338Z","shell.execute_reply":"2024-10-17T16:58:48.979048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.isnull().sum().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:58:48.982068Z","iopub.execute_input":"2024-10-17T16:58:48.982581Z","iopub.status.idle":"2024-10-17T16:58:49.851076Z","shell.execute_reply.started":"2024-10-17T16:58:48.982527Z","shell.execute_reply":"2024-10-17T16:58:49.849946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.isnull().sum().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:58:49.852730Z","iopub.execute_input":"2024-10-17T16:58:49.853202Z","iopub.status.idle":"2024-10-17T16:58:49.864095Z","shell.execute_reply.started":"2024-10-17T16:58:49.853151Z","shell.execute_reply":"2024-10-17T16:58:49.862951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate missing values\nmissing_values = train_data.isnull().mean() * 100\n\n# Plot\nmissing_values.plot(kind='bar', figsize=(25, 5), color='skyblue')\nplt.title('Train Data Percentage of Missing Values by Feature')\nplt.ylabel('Percentage')\nplt.xlabel('Features')\nplt.xticks(rotation=45)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:58:49.865700Z","iopub.execute_input":"2024-10-17T16:58:49.866149Z","iopub.status.idle":"2024-10-17T16:58:51.369367Z","shell.execute_reply.started":"2024-10-17T16:58:49.866082Z","shell.execute_reply":"2024-10-17T16:58:51.368221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate missing values\nmissing_values = test_data.isnull().mean() * 100\n\n# Plot\nmissing_values.plot(kind='bar', figsize=(25, 5), color='skyblue')\nplt.title('Test Data Percentage of Missing Values by Feature')\nplt.ylabel('Percentage')\nplt.xlabel('Features')\nplt.xticks(rotation=45)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:58:51.370863Z","iopub.execute_input":"2024-10-17T16:58:51.371862Z","iopub.status.idle":"2024-10-17T16:58:52.273190Z","shell.execute_reply.started":"2024-10-17T16:58:51.371811Z","shell.execute_reply":"2024-10-17T16:58:52.271946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_cols = ['responder_0','responder_1','responder_2','responder_3','responder_4','responder_5','responder_6','responder_7','responder_8']","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:58:52.274917Z","iopub.execute_input":"2024-10-17T16:58:52.275966Z","iopub.status.idle":"2024-10-17T16:58:52.280890Z","shell.execute_reply.started":"2024-10-17T16:58:52.275920Z","shell.execute_reply":"2024-10-17T16:58:52.279679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data[target_cols].hist(bins=30, figsize=(15, 10))\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:58:52.282627Z","iopub.execute_input":"2024-10-17T16:58:52.283588Z","iopub.status.idle":"2024-10-17T16:58:55.895286Z","shell.execute_reply.started":"2024-10-17T16:58:52.283537Z","shell.execute_reply":"2024-10-17T16:58:55.893970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Distribution of the data:\ntrain_data.drop(target_cols, axis=1).hist(figsize=(30,15),color = 'skyblue', edgecolor='black')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:58:55.896744Z","iopub.execute_input":"2024-10-17T16:58:55.897096Z","iopub.status.idle":"2024-10-17T16:59:20.292108Z","shell.execute_reply.started":"2024-10-17T16:58:55.897062Z","shell.execute_reply":"2024-10-17T16:59:20.290843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Distribution of the data:\ntest_data.hist(figsize=(30,15),color = 'skyblue', edgecolor='black')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:59:20.293828Z","iopub.execute_input":"2024-10-17T16:59:20.294235Z","iopub.status.idle":"2024-10-17T16:59:32.728418Z","shell.execute_reply.started":"2024-10-17T16:59:20.294194Z","shell.execute_reply":"2024-10-17T16:59:32.727217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_cols = list(train_data.select_dtypes(exclude=['object']).columns.difference(['responder_0','responder_1','responder_2','responder_3','responder_4','responder_5','responder_6','responder_7','responder_8','partition_id']))\ncat_cols = list(train_data.select_dtypes(include=['object']).columns)\n\nnum_cols_test = list(test_data.select_dtypes(exclude=['object']).columns.difference(['row_id','is_scored']))\ncat_cols_test = list(test_data.select_dtypes(include=['object']).columns)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:59:32.730049Z","iopub.execute_input":"2024-10-17T16:59:32.730557Z","iopub.status.idle":"2024-10-17T16:59:33.461354Z","shell.execute_reply.started":"2024-10-17T16:59:32.730503Z","shell.execute_reply":"2024-10-17T16:59:33.460144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 1: Impute numerical columns (e.g., fill missing values with the median)\ntrain_data[num_cols] = train_data[num_cols].fillna(train_data[num_cols].median())\ntest_data[num_cols] = test_data[num_cols].fillna(test_data[num_cols].median())\n\n# Step 2: Impute categorical columns (e.g., fill missing values with the mode or a custom value)\ntrain_data[cat_cols] = train_data[cat_cols].fillna('missing')\ntest_data[cat_cols] = test_data[cat_cols].fillna('missing')","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:59:33.466452Z","iopub.execute_input":"2024-10-17T16:59:33.466930Z","iopub.status.idle":"2024-10-17T16:59:46.581840Z","shell.execute_reply.started":"2024-10-17T16:59:33.466890Z","shell.execute_reply":"2024-10-17T16:59:46.580693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  object datatype columns encoding:\nfrom sklearn.preprocessing import LabelEncoder\nlabelencoder = LabelEncoder()\nfor col_name in cat_cols:\n    train_data[col_name]=labelencoder.fit_transform(train_data[col_name]).astype(int)\n        \n#for col_name in cat_cols_test:\n    test_data[col_name]=labelencoder.transform(test_data[col_name]).astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:59:46.583773Z","iopub.execute_input":"2024-10-17T16:59:46.584277Z","iopub.status.idle":"2024-10-17T16:59:46.590600Z","shell.execute_reply.started":"2024-10-17T16:59:46.584187Z","shell.execute_reply":"2024-10-17T16:59:46.589356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\nscaler = StandardScaler()\ntrain_data[num_cols] = scaler.fit_transform(train_data[num_cols])\ntest_data[num_cols] = scaler.transform(test_data[num_cols])","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:59:46.592634Z","iopub.execute_input":"2024-10-17T16:59:46.593099Z","iopub.status.idle":"2024-10-17T16:59:58.449269Z","shell.execute_reply.started":"2024-10-17T16:59:46.593047Z","shell.execute_reply":"2024-10-17T16:59:58.448283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX = train_data.drop(['partition_id','responder_0','responder_1','responder_2','responder_3','responder_4','responder_5','responder_6','responder_7','responder_8'],axis=1)\ny = train_data['responder_6']\ntest = test_data.drop(['row_id','is_scored'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:59:58.450820Z","iopub.execute_input":"2024-10-17T16:59:58.451300Z","iopub.status.idle":"2024-10-17T16:59:59.939891Z","shell.execute_reply.started":"2024-10-17T16:59:58.451246Z","shell.execute_reply":"2024-10-17T16:59:59.938586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape, y.shape, test.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:59:59.941479Z","iopub.execute_input":"2024-10-17T16:59:59.941871Z","iopub.status.idle":"2024-10-17T16:59:59.949736Z","shell.execute_reply.started":"2024-10-17T16:59:59.941833Z","shell.execute_reply":"2024-10-17T16:59:59.948472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(type(X))\nprint(type(y))","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:59:59.951429Z","iopub.execute_input":"2024-10-17T16:59:59.951848Z","iopub.status.idle":"2024-10-17T16:59:59.960614Z","shell.execute_reply.started":"2024-10-17T16:59:59.951807Z","shell.execute_reply":"2024-10-17T16:59:59.959440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from xgboost import XGBRegressor\nfrom sklearn.metrics import r2_score\nfrom sklearn.model_selection import train_test_split\n\n# Step 1: Split the data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Step 2: Initialize the XGBRegressor model\nmodel = XGBRegressor(n_estimators=100, learning_rate=0.1, max_depth=5, random_state=42)\n\n# Step 3: Fit the model on the training data\nmodel.fit(X_train, y_train)\n\n# Step 4: Make predictions on the test set\ny_pred = model.predict(X_test)\n#pred = model.predict(test)\n\n# Step 5: Calculate R² score\nr2 = r2_score(y_test, y_pred)\n\n# Print the R² score\nprint(f\"R² score: {r2}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:59:59.962153Z","iopub.execute_input":"2024-10-17T16:59:59.962648Z","iopub.status.idle":"2024-10-17T17:01:49.137061Z","shell.execute_reply.started":"2024-10-17T16:59:59.962596Z","shell.execute_reply":"2024-10-17T17:01:49.135631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom xgboost import XGBRegressor\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import r2_score\n\n# Initialize KFold\nkf = KFold(n_splits=5, shuffle=True, random_state=42)\n\n# Initialize lists to store results\nr2_scores = []\npreds = []\n\n# Perform K-Fold cross-validation\nfor train_index, test_index in kf.split(X):\n    X_train, X_test = X.iloc[train_index], X.iloc[test_index]\n    y_train, y_test = y.iloc[train_index], y.iloc[test_index]\n    \n    # Train XGBRegressor\n    model = XGBRegressor(n_estimators=100, learning_rate=0.1, max_depth=5, random_state=42)\n    model.fit(X_train, y_train)\n   \n    # Predict on the test set\n    y_pred = model.predict(X_test)\n    preds.append(model.predict(test))  # Assuming 'test' is your test DataFrame\n    \n    # Evaluate the model\n    r2 = r2_score(y_test, y_pred)\n    r2_scores.append(r2)\n\n# Calculate the mean R² score\nmean_r2 = np.mean(r2_scores)\n\n# Print results\nprint(f'R² scores for each fold: {r2_scores}')\nprint(f'Mean R²: {mean_r2}')","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:01:49.138607Z","iopub.execute_input":"2024-10-17T17:01:49.139051Z","iopub.status.idle":"2024-10-17T17:10:39.695420Z","shell.execute_reply.started":"2024-10-17T17:01:49.138954Z","shell.execute_reply":"2024-10-17T17:10:39.693957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({'row_id': sample_data.row_id, 'responder_6': np.mean(preds, axis=0)})\n# Save the DataFrame as a Parquet file\nsubmission.to_parquet('submission.parquet', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:10:39.697205Z","iopub.execute_input":"2024-10-17T17:10:39.697636Z","iopub.status.idle":"2024-10-17T17:10:39.737580Z","shell.execute_reply.started":"2024-10-17T17:10:39.697595Z","shell.execute_reply":"2024-10-17T17:10:39.736124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"from sklearn.model_selection import GridSearchCV\n\n# Parameter grid for XGBRegressor\nparam_grid = {\n    'n_estimators': [50, 100, 200],\n    'learning_rate': [0.01, 0.1, 0.2],\n    'max_depth': [3, 5, 7]\n}\n\n# GridSearchCV with 5-fold cross-validation and R² score\ngrid_search = GridSearchCV(XGBRegressor(random_state=42), param_grid, cv=kf, scoring='r2', n_jobs=-1)\n\n# Fit the model with GridSearchCV\ngrid_search.fit(X, y)\n\n# Best parameters and best R² score\nprint(f\"Best parameters: {grid_search.best_params_}\")\nprint(f\"Best R²: {grid_search.best_score_}\")","metadata":{}}]}