{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Importing Libraries","metadata":{}},{"cell_type":"code","source":"import optuna\nimport pandas as pd\nimport numpy as np\nimport warnings\nfrom sklearn import preprocessing\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split, StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.metrics import mean_squared_error\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nfrom scipy.optimize import minimize\nfrom statsmodels.stats.outliers_influence import variance_inflation_factor\n      \nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import StackingRegressor\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.linear_model import ElasticNet\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.ensemble import ExtraTreesRegressor\nfrom sklearn.ensemble import HistGradientBoostingRegressor\nfrom sklearn.svm import SVR\nfrom sklearn.neighbors import KNeighborsRegressor\n\nSEED = 42","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:12:56.584094Z","iopub.execute_input":"2024-11-14T19:12:56.584827Z","iopub.status.idle":"2024-11-14T19:13:02.591633Z","shell.execute_reply.started":"2024-11-14T19:12:56.584784Z","shell.execute_reply":"2024-11-14T19:13:02.590849Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load the data","metadata":{}},{"cell_type":"code","source":"# Define the file paths\ntrain_csv_path = '/kaggle/input/child-mind-institute-problematic-internet-use/train.csv'\ntest_csv_path = '/kaggle/input/child-mind-institute-problematic-internet-use/test.csv'\ndata_dictionary_path = '/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv'\nactigraphy_train_path = '/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet'\nactigraphy_test_path = '/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet'\n\n# Load the tabular data\ntrain_data = pd.read_csv(train_csv_path)\ntest_data = pd.read_csv(test_csv_path)\ndata_dictionary = pd.read_csv(data_dictionary_path)","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:13:05.379862Z","iopub.execute_input":"2024-11-14T19:13:05.381039Z","iopub.status.idle":"2024-11-14T19:13:05.467139Z","shell.execute_reply.started":"2024-11-14T19:13:05.380992Z","shell.execute_reply":"2024-11-14T19:13:05.466192Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dictionary.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:13:11.956971Z","iopub.execute_input":"2024-11-14T19:13:11.957839Z","iopub.status.idle":"2024-11-14T19:13:11.976622Z","shell.execute_reply.started":"2024-11-14T19:13:11.957800Z","shell.execute_reply":"2024-11-14T19:13:11.975809Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:13:13.269737Z","iopub.execute_input":"2024-11-14T19:13:13.270615Z","iopub.status.idle":"2024-11-14T19:13:13.304057Z","shell.execute_reply.started":"2024-11-14T19:13:13.270573Z","shell.execute_reply":"2024-11-14T19:13:13.303086Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:13:14.229026Z","iopub.execute_input":"2024-11-14T19:13:14.229905Z","iopub.status.idle":"2024-11-14T19:13:14.255016Z","shell.execute_reply.started":"2024-11-14T19:13:14.229862Z","shell.execute_reply":"2024-11-14T19:13:14.254116Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Function Definitions for Processing Time Series Data from Parquet Files","metadata":{}},{"cell_type":"code","source":"import os\n# Custom functions\ndef process_file(filename, dirname):\n    data = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    data.drop('step', axis=1, inplace=True)\n    return data.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname):\n    ids = os.listdir(dirname)\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    stats, indexes = zip(*results)\n    data = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    data['id'] = indexes\n    return data","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:13:16.398608Z","iopub.execute_input":"2024-11-14T19:13:16.398981Z","iopub.status.idle":"2024-11-14T19:13:16.406833Z","shell.execute_reply.started":"2024-11-14T19:13:16.398944Z","shell.execute_reply":"2024-11-14T19:13:16.405868Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# `load_time_series` Function Overview\n\nThe `load_time_series` function is designed to load multiple time-series files from a specified directory and process them in parallel, resulting in a summarized DataFrame. Below is a detailed breakdown of its functionality:\n\n\n## 1. Retrieve All File/Folder Names\n\n    * os.listdir(dirname) gets a list of all files or folders in the specified dirname directory and stores it in ids.\n\n## 2. Process Files in Parallel:\n\n    * ThreadPoolExecutor() is used to process the files concurrently. This parallel processing speeds up the function, especially if there are many files to process.\n\n    * For each file/folder name in ids, it calls the process_file function using a lambda function, with fname as the filename and dirname as the directory.\n\n    * tqdm() is used to display a progress bar, showing how many files have been processed out of the total (total=len(ids)).\n\n## 3. Unpack the Results:\n\n* results is a list where each entry contains:\n    * The flattened summary statistics of each file (from process_file).\n    * An identifier derived from the filename (also from process_file).\n\n* zip(results) splits this list of tuples into two separate lists:\n    * stats holds the summary statistics arrays for each file.\n    * indexes holds the identifiers (extracted from the filename).\n\n## 4. Create a DataFrame from Summary Statistics:\n\n    * A DataFrame data is created with stats as rows, with each column named stat_0, stat_1, etc., based on the number of summary statistics.\n\n    * data['id'] = indexes adds an id column with unique identifiers from each file.\n\n## 5. Return the Summary DataFrame:\n\n    * Finally, the function returns data, a DataFrame where each row represents a summarized time-series file with its unique identifier.\n\nThis function is efficient for summarizing large numbers of time-series files quickly and storing the summary information in a single, structured DataFrame.","metadata":{}},{"cell_type":"code","source":"train_parquet = load_time_series(actigraphy_train_path)\ntest_parquet = load_time_series(actigraphy_test_path)","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:13:18.817286Z","iopub.execute_input":"2024-11-14T19:13:18.817916Z","iopub.status.idle":"2024-11-14T19:14:39.476758Z","shell.execute_reply.started":"2024-11-14T19:13:18.817877Z","shell.execute_reply":"2024-11-14T19:14:39.475846Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Merge and preprocess data\ntrain_df = pd.merge(train_data, train_parquet, how=\"left\", on='id')\ntest_df = pd.merge(test_data, test_parquet, how=\"left\", on='id')","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:15:13.673600Z","iopub.execute_input":"2024-11-14T19:15:13.674475Z","iopub.status.idle":"2024-11-14T19:15:13.701257Z","shell.execute_reply.started":"2024-11-14T19:15:13.674435Z","shell.execute_reply":"2024-11-14T19:15:13.700493Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:15:14.675748Z","iopub.execute_input":"2024-11-14T19:15:14.676140Z","iopub.status.idle":"2024-11-14T19:15:14.682401Z","shell.execute_reply.started":"2024-11-14T19:15:14.676104Z","shell.execute_reply":"2024-11-14T19:15:14.681404Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T19:15:15.643880Z","iopub.execute_input":"2024-11-14T19:15:15.644614Z","iopub.status.idle":"2024-11-14T19:15:15.669898Z","shell.execute_reply.started":"2024-11-14T19:15:15.644577Z","shell.execute_reply":"2024-11-14T19:15:15.668924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:15:16.730381Z","iopub.execute_input":"2024-11-14T19:15:16.731265Z","iopub.status.idle":"2024-11-14T19:15:16.737932Z","shell.execute_reply.started":"2024-11-14T19:15:16.731214Z","shell.execute_reply":"2024-11-14T19:15:16.736978Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop all columns starting with 'PCIAT-PCIAT' except 'PCIAT-PCIAT_Total'\ncolumns_to_keep = ['PCIAT-PCIAT_Total']\ncolumns_to_drop = [col for col in train_df.columns if col.startswith('PCIAT-PCIAT') and col != 'PCIAT-PCIAT_Total']\n\n# Drop the columns\ntrain_df = train_df.drop(columns=columns_to_drop)\n\nprint(f\"Columns dropped: {columns_to_drop}\")\nprint(f\"Remaining columns: {train_df.columns}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:15:17.714023Z","iopub.execute_input":"2024-11-14T19:15:17.714422Z","iopub.status.idle":"2024-11-14T19:15:17.724836Z","shell.execute_reply.started":"2024-11-14T19:15:17.714386Z","shell.execute_reply":"2024-11-14T19:15:17.723889Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# List of columns in the test dataset\ntest_columns = test_df.columns.tolist()\n\n# List of columns in the train dataset \ntrain_columns = train_df.columns.tolist()\n\n# Find columns that are in the train dataset but not in the test dataset\ncolumns_only_in_train = list(set(train_columns) - set(test_columns))\n\n# Find columns that are in the test dataset but not in the train dataset\ncolumns_only_in_test = list(set(test_columns) - set(train_columns))\n\n# Display the non-matching columns\nprint(\"Columns only in the train dataset:\", columns_only_in_train)\nprint(\"Columns only in the test dataset:\", columns_only_in_test)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:15:18.676491Z","iopub.execute_input":"2024-11-14T19:15:18.677301Z","iopub.status.idle":"2024-11-14T19:15:18.683933Z","shell.execute_reply.started":"2024-11-14T19:15:18.677251Z","shell.execute_reply":"2024-11-14T19:15:18.682899Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Dropping the 'PCIAT-PCIAT_Total', 'PCIAT-Season' columns, as we need to predict 'sii' which is derived from 'PCIAT-PCIAT_Total' and both the above columns are not present in test data set.**","metadata":{}},{"cell_type":"code","source":"# Drop the extra columns from the train dataset\ncolumns_to_drop = ['PCIAT-PCIAT_Total', 'PCIAT-Season']\ntrain_df = train_df.drop(columns=columns_to_drop)\n\n# Check the updated columns in train\nprint(train_df.columns)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:15:20.527283Z","iopub.execute_input":"2024-11-14T19:15:20.527911Z","iopub.status.idle":"2024-11-14T19:15:20.537706Z","shell.execute_reply.started":"2024-11-14T19:15:20.527872Z","shell.execute_reply":"2024-11-14T19:15:20.536836Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Creating a mask for missing values\nmissing_values_mask = train_df.isnull()\n\n# Setting up the matplotlib figure\nplt.figure(figsize=(12, 8))\n\n# Creating th heatmap\nsns.heatmap(missing_values_mask, cbar=False, cmap='viridis', yticklabels=False)\n\n# Customizing the plot\nplt.title('Missing Values Visualization for train_df', fontsize=16)\nplt.xlabel('Features', fontsize=14)\nplt.ylabel('Samples', fontsize=14)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:15:21.597812Z","iopub.execute_input":"2024-11-14T19:15:21.598203Z","iopub.status.idle":"2024-11-14T19:15:22.937808Z","shell.execute_reply.started":"2024-11-14T19:15:21.598164Z","shell.execute_reply":"2024-11-14T19:15:22.936882Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T19:15:24.467799Z","iopub.execute_input":"2024-11-14T19:15:24.468910Z","iopub.status.idle":"2024-11-14T19:15:24.475061Z","shell.execute_reply.started":"2024-11-14T19:15:24.468869Z","shell.execute_reply":"2024-11-14T19:15:24.474081Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T19:15:25.467713Z","iopub.execute_input":"2024-11-14T19:15:25.468115Z","iopub.status.idle":"2024-11-14T19:15:25.474340Z","shell.execute_reply.started":"2024-11-14T19:15:25.468073Z","shell.execute_reply":"2024-11-14T19:15:25.473450Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_columns = train_df.select_dtypes(include=['object', 'category']).columns.tolist()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:15:26.729479Z","iopub.execute_input":"2024-11-14T19:15:26.730145Z","iopub.status.idle":"2024-11-14T19:15:26.735898Z","shell.execute_reply.started":"2024-11-14T19:15:26.730095Z","shell.execute_reply":"2024-11-14T19:15:26.734805Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_columns","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:15:28.593132Z","iopub.execute_input":"2024-11-14T19:15:28.593981Z","iopub.status.idle":"2024-11-14T19:15:28.599873Z","shell.execute_reply.started":"2024-11-14T19:15:28.593939Z","shell.execute_reply":"2024-11-14T19:15:28.598880Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Filling missing values in categorical columns with mode\nfor col in categorical_columns:\n    train_df[col]= train_df[col].fillna(train_df[col].mode()[0])\n    test_df[col]= test_df[col].fillna(test_df[col].mode()[0])","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:15:29.670640Z","iopub.execute_input":"2024-11-14T19:15:29.671013Z","iopub.status.idle":"2024-11-14T19:15:29.703666Z","shell.execute_reply.started":"2024-11-14T19:15:29.670978Z","shell.execute_reply":"2024-11-14T19:15:29.702673Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T19:15:30.657537Z","iopub.execute_input":"2024-11-14T19:15:30.657920Z","iopub.status.idle":"2024-11-14T19:15:30.664171Z","shell.execute_reply.started":"2024-11-14T19:15:30.657883Z","shell.execute_reply":"2024-11-14T19:15:30.663255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T19:15:31.826906Z","iopub.execute_input":"2024-11-14T19:15:31.827597Z","iopub.status.idle":"2024-11-14T19:15:31.833207Z","shell.execute_reply.started":"2024-11-14T19:15:31.827556Z","shell.execute_reply":"2024-11-14T19:15:31.832099Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"non_categorical_columns = train_df.select_dtypes(include=['float64', 'int64']).columns","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:15:32.758552Z","iopub.execute_input":"2024-11-14T19:15:32.758926Z","iopub.status.idle":"2024-11-14T19:15:32.767658Z","shell.execute_reply.started":"2024-11-14T19:15:32.758890Z","shell.execute_reply":"2024-11-14T19:15:32.766727Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"non_categorical_columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T19:15:33.646848Z","iopub.execute_input":"2024-11-14T19:15:33.647610Z","iopub.status.idle":"2024-11-14T19:15:33.653706Z","shell.execute_reply.started":"2024-11-14T19:15:33.647571Z","shell.execute_reply":"2024-11-14T19:15:33.652822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Filling missing values with the mean for numerical columns\nfor col in non_categorical_columns:\n    # Fill missing values with the mean and reassign to the DataFrame\n    train_df[col] = train_df[col].fillna(train_df[col].mean())  # For training data\n\n    # Check if the column exists in the test DataFrame before filling\n    if col in test_df.columns:\n        test_df[col] = test_df[col].fillna(test_df[col].mean())\n    else:\n        print(f\"Column '{col}' not found in test_df. Skipping...\")\n\n# Check for any remaining missing values\nprint(\"Missing values in train_df after filling:\\n\", train_df.isnull().sum())\nprint(\"Missing values in test_df after filling:\\n\", test_df.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:15:34.988336Z","iopub.execute_input":"2024-11-14T19:15:34.989199Z","iopub.status.idle":"2024-11-14T19:15:35.115643Z","shell.execute_reply.started":"2024-11-14T19:15:34.989159Z","shell.execute_reply":"2024-11-14T19:15:35.114721Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T19:15:36.155611Z","iopub.execute_input":"2024-11-14T19:15:36.156458Z","iopub.status.idle":"2024-11-14T19:15:36.161956Z","shell.execute_reply.started":"2024-11-14T19:15:36.156419Z","shell.execute_reply":"2024-11-14T19:15:36.161105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Creating a mask for missing values\nmissing_values_mask = train_df.isnull()\n\n# Setting up the matplotlib figure\nplt.figure(figsize=(12, 8))\n\n# Creating th heatmap\nsns.heatmap(missing_values_mask, cbar=False, cmap='viridis', yticklabels=False)\n\n# Customizing the plot\nplt.title('Missing Values Visualization for train_df', fontsize=16)\nplt.xlabel('Features', fontsize=14)\nplt.ylabel('Samples', fontsize=14)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T19:15:41.125275Z","iopub.execute_input":"2024-11-14T19:15:41.126005Z","iopub.status.idle":"2024-11-14T19:15:42.157208Z","shell.execute_reply.started":"2024-11-14T19:15:41.125968Z","shell.execute_reply":"2024-11-14T19:15:42.156360Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df=train_df.dropna(subset=['sii'])\ntrain_df.reset_index(drop=True, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T19:15:47.691697Z","iopub.execute_input":"2024-11-14T19:15:47.692618Z","iopub.status.idle":"2024-11-14T19:15:47.706728Z","shell.execute_reply.started":"2024-11-14T19:15:47.692576Z","shell.execute_reply":"2024-11-14T19:15:47.705780Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T19:15:48.399830Z","iopub.execute_input":"2024-11-14T19:15:48.400765Z","iopub.status.idle":"2024-11-14T19:15:48.407355Z","shell.execute_reply.started":"2024-11-14T19:15:48.400716Z","shell.execute_reply":"2024-11-14T19:15:48.406321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.preprocessing import OneHotEncoder\n\n# Identify categorical columns and ensure their dtype is 'object' or 'category'\ncategorical_columns = train_df.select_dtypes(include=['object', 'category']).columns.tolist()\n\n# Initialize OneHotEncoder with handle_unknown to ignore unseen categories in test data\nencoder = OneHotEncoder(handle_unknown='ignore', sparse_output=False)\n\n# Fit the encoder on the categorical columns of the training data\nencoder.fit(train_df[categorical_columns])\n\n# Transform both train and test data\ntrain_encoded = pd.DataFrame(encoder.transform(train_df[categorical_columns]), \n                             columns=encoder.get_feature_names_out(categorical_columns))\n\ntest_encoded = pd.DataFrame(encoder.transform(test_df[categorical_columns]), \n                            columns=encoder.get_feature_names_out(categorical_columns))\n\n# Add back non-categorical columns (if any) from the original datasets\ndf_train_encoded = pd.concat([train_df.drop(categorical_columns, axis=1).reset_index(drop=True), \n                              train_encoded.reset_index(drop=True)], axis=1)\n\ndf_test_encoded = pd.concat([test_df.drop(categorical_columns, axis=1).reset_index(drop=True), \n                             test_encoded.reset_index(drop=True)], axis=1)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:15:54.896798Z","iopub.execute_input":"2024-11-14T19:15:54.897552Z","iopub.status.idle":"2024-11-14T19:15:55.261955Z","shell.execute_reply.started":"2024-11-14T19:15:54.897509Z","shell.execute_reply":"2024-11-14T19:15:55.260974Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train_encoded.shape","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:15:56.705377Z","iopub.execute_input":"2024-11-14T19:15:56.706240Z","iopub.status.idle":"2024-11-14T19:15:56.711965Z","shell.execute_reply.started":"2024-11-14T19:15:56.706199Z","shell.execute_reply":"2024-11-14T19:15:56.711007Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test_encoded.shape","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:16:00.176235Z","iopub.execute_input":"2024-11-14T19:16:00.176631Z","iopub.status.idle":"2024-11-14T19:16:00.182762Z","shell.execute_reply.started":"2024-11-14T19:16:00.176594Z","shell.execute_reply":"2024-11-14T19:16:00.181797Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# LGBMClassifier Hyperparameter Tuning using Optuna","metadata":{}},{"cell_type":"code","source":"import optuna\nfrom sklearn.model_selection import cross_val_score, StratifiedKFold\nfrom lightgbm import LGBMClassifier\nfrom sklearn.metrics import make_scorer, cohen_kappa_score\nimport numpy as np\n\n# Define custom QWK scorer\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n# Custom scorer for Optuna\nKAPPA_SCORER = make_scorer(quadratic_weighted_kappa, greater_is_better=True)\n\n# Define the objective function for Optuna\ndef lgb_objective(trial):\n    params = {\n        'objective':         'multiclass',\n        'num_class':         4,  # For 4 classes: 0, 1, 2, 3\n        'verbosity':         -1,\n        'random_state':      SEED,\n        'boosting_type':     'gbdt',\n        'device': 'gpu',\n        'lambda_l1':         trial.suggest_float('lambda_l1', 1e-3, 10.0, log=True),\n        'lambda_l2':         trial.suggest_float('lambda_l2', 1e-3, 10.0, log=True),\n        'learning_rate':     trial.suggest_float('learning_rate', 1e-2, 1e-1, log=True),\n        'max_depth':         trial.suggest_int('max_depth', 4, 8),\n        'num_leaves':        trial.suggest_int('num_leaves', 16, 256),\n        'colsample_bytree':  trial.suggest_float('colsample_bytree', 0.4, 1.0),\n        'colsample_bynode':  trial.suggest_float('colsample_bynode', 0.4, 1.0),\n        'bagging_fraction':  trial.suggest_float('bagging_fraction', 0.4, 1.0),\n        'bagging_freq':      trial.suggest_int('bagging_freq', 1, 7),\n        'min_data_in_leaf':  trial.suggest_int('min_data_in_leaf', 5, 100),\n    }\n\n    # Feature and target columns\n    X = df_train_encoded.drop(columns=['sii'])\n    y = df_train_encoded['sii'].astype('int')\n\n    # StratifiedKFold cross-validation\n    cv = StratifiedKFold(n_splits=5, shuffle=True, random_state=SEED)\n    model = LGBMClassifier(**params)\n\n    # Cross-validation scoring\n    val_scores = cross_val_score(\n        estimator=model, \n        X=X, y=y, \n        cv=cv, \n        scoring=KAPPA_SCORER,  # QWK Scorer\n    )\n\n    # Return mean validation score\n    return np.mean(val_scores)\n\n# Optuna study to optimize the objective function\nstudy = optuna.create_study(direction='maximize')\nstudy.optimize(lgb_objective, n_trials=50)\n\n# Best hyperparameters\nprint(\"Best hyperparameters:\", study.best_params)","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:21:08.026755Z","iopub.execute_input":"2024-11-14T19:21:08.027429Z","iopub.status.idle":"2024-11-14T19:28:10.377523Z","shell.execute_reply.started":"2024-11-14T19:21:08.027390Z","shell.execute_reply":"2024-11-14T19:28:10.376593Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"study.best_params","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:30:16.997909Z","iopub.execute_input":"2024-11-14T19:30:16.998741Z","iopub.status.idle":"2024-11-14T19:30:17.006390Z","shell.execute_reply.started":"2024-11-14T19:30:16.998685Z","shell.execute_reply":"2024-11-14T19:30:17.005436Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from lightgbm import LGBMClassifier\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nimport numpy as np\n\n# Define the best hyperparameters for LGBMClassifier\nbest_lgb_params = study.best_params\n\n# Instantiate the classifier with best parameters\nlgb_classifier = LGBMClassifier(**best_lgb_params, verbosity=-1)\n\n# Assuming df_train_encoded has been preprocessed and 'sii' is the target column\nX = df_train_encoded.drop(columns=['sii'])  # Features\ny = df_train_encoded['sii'].astype('int')   # Target column converted to integers\n\n# Stratified K-Fold\nskf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# Store results\nfold_scores = []\n\n# Perform Stratified K-Fold Cross-Validation\nfor train_index, val_index in skf.split(X, y):\n    X_train, X_val = X.iloc[train_index], X.iloc[val_index]\n    y_train, y_val = y.iloc[train_index], y.iloc[val_index]\n\n    # Train the classifier\n    lgb_classifier.fit(X_train, y_train)\n\n    # Predict on the validation set\n    y_pred = lgb_classifier.predict(X_val)\n\n    # Calculate QWK (Quadratic Weighted Kappa)\n    qwk_score = cohen_kappa_score(y_val, y_pred, weights='quadratic')\n    fold_scores.append(qwk_score)\n\n# Average score across folds\naverage_qwk = np.mean(fold_scores)\n\n# Print results\nprint(f\"LGBMClassifier: QWK = {average_qwk:.4f}\")","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:30:33.247311Z","iopub.execute_input":"2024-11-14T19:30:33.247676Z","iopub.status.idle":"2024-11-14T19:30:38.149238Z","shell.execute_reply.started":"2024-11-14T19:30:33.247644Z","shell.execute_reply":"2024-11-14T19:30:38.148313Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predict on test data\ny_pred_test = lgb_classifier.predict(df_test_encoded)  # Test set predictions\n\n# Create submission file\nsubmission = pd.DataFrame({\n    'id': test_df['id'],  # Assuming there's an 'id' column in test set\n    'predicted_target': y_pred_test.astype(int)  # Ensure predictions are integers\n})\n\n# Save to CSV\nsubmission.to_csv('submission.csv', index=False)\nprint(\"Submission file created.\")","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:30:45.464674Z","iopub.execute_input":"2024-11-14T19:30:45.465062Z","iopub.status.idle":"2024-11-14T19:30:45.488856Z","shell.execute_reply.started":"2024-11-14T19:30:45.465014Z","shell.execute_reply":"2024-11-14T19:30:45.487759Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2024-11-14T19:30:49.708778Z","iopub.execute_input":"2024-11-14T19:30:49.709450Z","iopub.status.idle":"2024-11-14T19:30:49.719612Z","shell.execute_reply.started":"2024-11-14T19:30:49.709411Z","shell.execute_reply":"2024-11-14T19:30:49.718651Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# XGBClassifier Hyperparameter Tuning with Optuna:","metadata":{}},{"cell_type":"code","source":"'''import optuna\nfrom xgboost import XGBClassifier\nfrom sklearn.model_selection import cross_val_score, StratifiedKFold\nfrom sklearn.metrics import make_scorer, cohen_kappa_score\nimport numpy as np\n\n# Custom QWK scorer\ndef qwk_scorer(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights=\"quadratic\")\n\nKAPPA_SCORER = make_scorer(qwk_scorer, greater_is_better=True)\n\n# Optuna objective for XGBClassifier\ndef xgb_objective(trial):\n    params = {\n        'objective': 'multi:softmax',  # For multiclass classification\n        'num_class': len(np.unique(y)),  # Adjust for the number of classes\n        'device': 'cuda',\n        'lambda': trial.suggest_float('lambda', 1e-3, 10.0, log=True),\n        'alpha': trial.suggest_float('alpha', 1e-3, 10.0, log=True),\n        'learning_rate': trial.suggest_float('learning_rate', 1e-3, 0.3, log=True),\n        'n_estimators': trial.suggest_int('n_estimators', 100, 500),\n        'max_depth': trial.suggest_int('max_depth', 3, 10),\n        'min_child_weight': trial.suggest_float('min_child_weight', 1e-2, 10.0, log=True),\n        'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0)\n    }\n    \n    model = XGBClassifier(**params)\n    \n    # Perform StratifiedKFold cross-validation\n    skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n    scores = cross_val_score(model, X, y, cv=skf, scoring=KAPPA_SCORER)\n    \n    return np.mean(scores)\n\n# Optuna study for XGBClassifier\nxgb_study = optuna.create_study(direction='maximize')\nxgb_study.optimize(xgb_objective, n_trials=50)\n\n# Best hyperparameters\nprint(\"Best hyperparameters for XGBClassifier:\", xgb_study.best_params)'''","metadata":{"execution":{"iopub.status.busy":"2024-11-14T16:57:23.431716Z","iopub.status.idle":"2024-11-14T16:57:23.432182Z","shell.execute_reply.started":"2024-11-14T16:57:23.431938Z","shell.execute_reply":"2024-11-14T16:57:23.431963Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''best_xgb_hyperparameter={'lambda': 0.11701958586457006, 'alpha': 0.06199347062282436, 'learning_rate': 0.03116101640319547, 'n_estimators': 411, 'max_depth': 3, 'min_child_weight': 0.029987305329202723, 'subsample': 0.9252229830303276, 'colsample_bytree': 0.5612199607669004}'''","metadata":{"execution":{"iopub.status.busy":"2024-11-14T16:57:23.433390Z","iopub.status.idle":"2024-11-14T16:57:23.433725Z","shell.execute_reply.started":"2024-11-14T16:57:23.433555Z","shell.execute_reply":"2024-11-14T16:57:23.433574Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''from xgboost import XGBClassifier\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nimport numpy as np\n\n# Define the best hyperparameters for XGBClassifier\nbest_xgb_params = best_xgb_hyperparameter\n\n# Instantiate the classifier with best parameters\nxgb_classifier = XGBClassifier(**best_xgb_params, use_label_encoder=False, eval_metric='mlogloss')\n\n# Assuming df_train_encoded has been preprocessed and 'sii' is the target column\nX = df_train_encoded.drop(columns=['sii'])  # Features\ny = df_train_encoded['sii'].astype('int')   # Target column converted to integers\n\n# Stratified K-Fold\nskf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# Store results\nfold_scores = []\n\n# Perform Stratified K-Fold Cross-Validation\nfor train_index, val_index in skf.split(X, y):\n    X_train, X_val = X.iloc[train_index], X.iloc[val_index]\n    y_train, y_val = y.iloc[train_index], y.iloc[val_index]\n\n    # Train the classifier\n    xgb_classifier.fit(X_train, y_train)\n\n    # Predict on the validation set\n    y_pred = xgb_classifier.predict(X_val)\n\n    # Calculate QWK (Quadratic Weighted Kappa)\n    qwk_score = cohen_kappa_score(y_val, y_pred, weights='quadratic')\n    fold_scores.append(qwk_score)\n\n# Average score across folds\naverage_qwk = np.mean(fold_scores)\n\n# Print results\nprint(f\"XGBClassifier: QWK = {average_qwk:.4f}\")\n'''","metadata":{"execution":{"iopub.status.busy":"2024-11-14T16:57:23.435100Z","iopub.status.idle":"2024-11-14T16:57:23.435486Z","shell.execute_reply.started":"2024-11-14T16:57:23.435299Z","shell.execute_reply":"2024-11-14T16:57:23.435325Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''# Predict on test data\ny_pred_test = xgb_classifier.predict(df_test_encoded)  # Test set predictions\n\n# Create submission file\nsubmission = pd.DataFrame({\n    'id': test_df_cleaned['id'],  # Assuming there's an 'id' column in test set\n    'predicted_target': y_pred_test.astype(int)  # Ensure predictions are integers\n})\n\n# Save to CSV\nsubmission.to_csv('submission.csv', index=False)\nprint(\"Submission file created.\")'''","metadata":{"execution":{"iopub.status.busy":"2024-11-14T16:57:23.436591Z","iopub.status.idle":"2024-11-14T16:57:23.436933Z","shell.execute_reply.started":"2024-11-14T16:57:23.436760Z","shell.execute_reply":"2024-11-14T16:57:23.436779Z"},"trusted":true},"outputs":[],"execution_count":null}]}