{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### CMI: Problematic Internet use Prediction (timeseries and behavioural)","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport optuna\nimport pandas as pd\nimport os\nimport re\nimport tensorflow as tf\nimport torch\n\nfrom IPython.display import clear_output\nfrom concurrent.futures import ThreadPoolExecutor\nfrom datetime import datetime\nfrom colorama import Fore, Style\nfrom pyprojroot import here\nfrom scipy.stats import kurtosis\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom tqdm import tqdm\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor\n\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.impute import SimpleImputer, KNNImputer\n\npd.set_option('display.max_rows', None)\npd.set_option(\"display.max_columns\", None)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:19:39.669650Z","iopub.execute_input":"2024-11-23T22:19:39.670133Z","iopub.status.idle":"2024-11-23T22:19:39.685452Z","shell.execute_reply.started":"2024-11-23T22:19:39.670089Z","shell.execute_reply":"2024-11-23T22:19:39.684296Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train_b = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ndf_test_b = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ndf_sample_b = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ndf_train_b.head(5)\nprint(\"\\n\")\n\nprint(f\"df_train shape: {df_train_b.shape}\") # (3960, 82)\nprint(f\"df_test shape: {df_test_b.shape}\") # (20, 59)\nprint(f\"df_sample shape: {df_sample_b.shape}\") # (20, 2)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:10:34.344261Z","iopub.execute_input":"2024-11-23T22:10:34.344678Z","iopub.status.idle":"2024-11-23T22:10:34.402387Z","shell.execute_reply.started":"2024-11-23T22:10:34.344642Z","shell.execute_reply":"2024-11-23T22:10:34.401244Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Calculate summary stats for timeseries columns","metadata":{}},{"cell_type":"code","source":"def process_file(filename, dirname):\n\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n\n    # Convert 'time_of_day' to datetime\n    df['time_of_day'] = pd.to_datetime(df['time_of_day'])\n    df.set_index('time_of_day', inplace=True)\n\n    # Daily aggregation\n    daily_features = df.resample('D').agg({\n\n        'X': ['mean', 'std', 'max', 'min', 'median', 'skew', lambda x: kurtosis(x, nan_policy='omit')],\n        'Y': ['mean', 'std', 'max', 'min', 'median', 'skew', lambda x: kurtosis(x, nan_policy='omit')],\n        'Z': ['mean', 'std', 'max', 'min', 'median', 'skew', lambda x: kurtosis(x, nan_policy='omit')],\n        'enmo': ['mean', 'std', 'max', 'min', 'median', 'skew', lambda x: kurtosis(x, nan_policy='omit')],\n        'light': ['mean', 'std'],\n        'battery_voltage': ['mean', 'std'],\n        'non-wear_flag': 'sum'\n\n    })\n\n\n    # Flatten the MultiIndex columns\n    daily_features.columns = ['_'.join(col).strip() for col in daily_features.columns.values]\n\n    # Add additional features\n    daily_features['activity_count'] = (df['enmo'] > 0.1).resample('D').sum()  # Adjust threshold as needed\n    daily_features['days_active'] = (df['non-wear_flag'] == 0).resample('D').sum()  # Count active days\n\n    \n    # Rate of change features\n    daily_features['enmo_change'] = daily_features['enmo_mean'].diff()\n\n    # Rolling statistics\n    rolling_windows = [5, 10, 15]  # Days\n    for window in rolling_windows:\n        daily_features[f'enmo_rolling_mean_{window}'] = daily_features['enmo_mean'].rolling(window=window).mean()\n        daily_features[f'enmo_rolling_std_{window}'] = daily_features['enmo_std'].rolling(window=window).std()\n\n    # Frequency domain features using FFT\n    for axis in ['X', 'Y', 'Z', 'enmo']:\n        freq_features = np.fft.fft(df[axis])\n        daily_features[f'{axis}_dominant_freq'] = np.abs(freq_features).argmax()\n\n    # Reset index to retain 'id'\n    daily_features.reset_index(inplace=True)\n\n    # Extract child ID from filename\n    child_id = filename.split('=')[1]\n    daily_features['id'] = child_id\n\n    return daily_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:12:11.358426Z","iopub.execute_input":"2024-11-23T22:12:11.358962Z","iopub.status.idle":"2024-11-23T22:12:11.371714Z","shell.execute_reply.started":"2024-11-23T22:12:11.358923Z","shell.execute_reply":"2024-11-23T22:12:11.370330Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- Load all subjects into a single dataframe","metadata":{}},{"cell_type":"code","source":"def load_time_series(dirname) -> pd.DataFrame:\n\n    ids = os.listdir(dirname)\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n\n    \n    # Concatenate all results into a single DataFrame\n    features_df = pd.concat(results, ignore_index=True)\n\n    return features_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:12:33.116083Z","iopub.execute_input":"2024-11-23T22:12:33.116470Z","iopub.status.idle":"2024-11-23T22:12:33.123983Z","shell.execute_reply.started":"2024-11-23T22:12:33.116435Z","shell.execute_reply":"2024-11-23T22:12:33.122340Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features_df = load_time_series('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:13:06.937649Z","iopub.execute_input":"2024-11-23T22:13:06.939089Z","iopub.status.idle":"2024-11-23T22:19:39.667323Z","shell.execute_reply.started":"2024-11-23T22:13:06.939035Z","shell.execute_reply":"2024-11-23T22:19:39.666061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features_test = load_time_series('/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:20:32.219066Z","iopub.execute_input":"2024-11-23T22:20:32.219588Z","iopub.status.idle":"2024-11-23T22:20:32.947598Z","shell.execute_reply.started":"2024-11-23T22:20:32.219544Z","shell.execute_reply":"2024-11-23T22:20:32.946487Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features_df.tail()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:20:34.422704Z","iopub.execute_input":"2024-11-23T22:20:34.423134Z","iopub.status.idle":"2024-11-23T22:20:34.460244Z","shell.execute_reply.started":"2024-11-23T22:20:34.423098Z","shell.execute_reply":"2024-11-23T22:20:34.458706Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- Drop NaN columns","metadata":{}},{"cell_type":"code","source":"features_df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:20:35.906457Z","iopub.execute_input":"2024-11-23T22:20:35.907063Z","iopub.status.idle":"2024-11-23T22:20:35.921658Z","shell.execute_reply.started":"2024-11-23T22:20:35.907003Z","shell.execute_reply":"2024-11-23T22:20:35.920162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"columns_to_drop = [\n    'enmo_change',\n    'enmo_rolling_mean_5',\n    'enmo_rolling_std_5',\n    'enmo_rolling_mean_10',\n    'enmo_rolling_std_10',\n    'enmo_rolling_mean_15',\n    'enmo_rolling_std_15'\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:20:36.720182Z","iopub.execute_input":"2024-11-23T22:20:36.720573Z","iopub.status.idle":"2024-11-23T22:20:36.726848Z","shell.execute_reply.started":"2024-11-23T22:20:36.720537Z","shell.execute_reply":"2024-11-23T22:20:36.725234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features_df.drop(columns=columns_to_drop, inplace=True)\nfeatures_test.drop(columns=columns_to_drop, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:20:50.885012Z","iopub.execute_input":"2024-11-23T22:20:50.885405Z","iopub.status.idle":"2024-11-23T22:20:50.894072Z","shell.execute_reply.started":"2024-11-23T22:20:50.885372Z","shell.execute_reply":"2024-11-23T22:20:50.892417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train_ts = features_df.copy()\ndf_test_ts = features_test.copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:21:00.727007Z","iopub.execute_input":"2024-11-23T22:21:00.727382Z","iopub.status.idle":"2024-11-23T22:21:00.735247Z","shell.execute_reply.started":"2024-11-23T22:21:00.727351Z","shell.execute_reply":"2024-11-23T22:21:00.734110Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test_ts.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:21:01.797193Z","iopub.execute_input":"2024-11-23T22:21:01.797602Z","iopub.status.idle":"2024-11-23T22:21:01.831289Z","shell.execute_reply.started":"2024-11-23T22:21:01.797565Z","shell.execute_reply":"2024-11-23T22:21:01.829691Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train_b.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:21:03.149613Z","iopub.execute_input":"2024-11-23T22:21:03.151212Z","iopub.status.idle":"2024-11-23T22:21:03.226256Z","shell.execute_reply.started":"2024-11-23T22:21:03.151166Z","shell.execute_reply":"2024-11-23T22:21:03.224888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.merge(df_train_b, df_train_ts, how=\"left\", on='id')\ndf_test = pd.merge(df_test_b, df_test_ts, how=\"left\", on='id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:21:04.863940Z","iopub.execute_input":"2024-11-23T22:21:04.864415Z","iopub.status.idle":"2024-11-23T22:21:04.899951Z","shell.execute_reply.started":"2024-11-23T22:21:04.864378Z","shell.execute_reply":"2024-11-23T22:21:04.898687Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:21:07.080334Z","iopub.execute_input":"2024-11-23T22:21:07.080842Z","iopub.status.idle":"2024-11-23T22:21:07.163014Z","shell.execute_reply.started":"2024-11-23T22:21:07.080784Z","shell.execute_reply":"2024-11-23T22:21:07.161659Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Impute data with 5 neighbors\n","metadata":{}},{"cell_type":"code","source":"imputer = KNNImputer(n_neighbors=5)\nnumeric_cols = df_train.select_dtypes(include=['float64', 'int64']).columns\nimputed_data = imputer.fit_transform(df_train[numeric_cols])\ndf_train_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ndf_train_imputed['sii'] = df_train_imputed['sii'].round().astype(int)\n\nfor col in df_train.columns:\n    if col not in numeric_cols:\n        df_train_imputed[col] = df_train[col]\n\ndf_train = df_train_imputed\n\ndf_train.columns\ndf_train.shape\ndf_train.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:21:09.096976Z","iopub.execute_input":"2024-11-23T22:21:09.097384Z","iopub.status.idle":"2024-11-23T22:21:19.744779Z","shell.execute_reply.started":"2024-11-23T22:21:09.097348Z","shell.execute_reply":"2024-11-23T22:21:19.743687Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train_ts.shape, df_train_b.shape, df_train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:21:21.107509Z","iopub.execute_input":"2024-11-23T22:21:21.108045Z","iopub.status.idle":"2024-11-23T22:21:21.116111Z","shell.execute_reply.started":"2024-11-23T22:21:21.108003Z","shell.execute_reply":"2024-11-23T22:21:21.114866Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nfeaturesCols = ['id', 'Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'X_mean', 'X_std', 'X_max', 'X_min', 'X_median',\n       'X_skew', 'X_<lambda_0>', 'Y_mean', 'Y_std', 'Y_max', 'Y_min',\n       'Y_median', 'Y_skew', 'Y_<lambda_0>', 'Z_mean', 'Z_std', 'Z_max',\n       'Z_min', 'Z_median', 'Z_skew', 'Z_<lambda_0>', 'enmo_mean', 'enmo_std',\n       'enmo_max', 'enmo_min', 'enmo_median', 'enmo_skew', 'enmo_<lambda_0>',\n       'light_mean', 'light_std', 'battery_voltage_mean',\n       'battery_voltage_std', 'non-wear_flag_sum', 'activity_count',\n       'days_active', 'X_dominant_freq', 'Y_dominant_freq', 'Z_dominant_freq',\n       'enmo_dominant_freq','sii',]\n\n\n\n\n\nfeaturesColsTest = ['id', 'Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'X_mean', 'X_std', 'X_max', 'X_min', 'X_median',\n       'X_skew', 'X_<lambda_0>', 'Y_mean', 'Y_std', 'Y_max', 'Y_min',\n       'Y_median', 'Y_skew', 'Y_<lambda_0>', 'Z_mean', 'Z_std', 'Z_max',\n       'Z_min', 'Z_median', 'Z_skew', 'Z_<lambda_0>', 'enmo_mean', 'enmo_std',\n       'enmo_max', 'enmo_min', 'enmo_median', 'enmo_skew', 'enmo_<lambda_0>',\n       'light_mean', 'light_std', 'battery_voltage_mean',\n       'battery_voltage_std', 'non-wear_flag_sum', 'activity_count',\n       'days_active', 'X_dominant_freq', 'Y_dominant_freq', 'Z_dominant_freq',\n       'enmo_dominant_freq',]\n\n# categorical cols\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season',\n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season',\n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:21:21.988659Z","iopub.execute_input":"2024-11-23T22:21:21.989588Z","iopub.status.idle":"2024-11-23T22:21:22.003040Z","shell.execute_reply.started":"2024-11-23T22:21:21.989506Z","shell.execute_reply":"2024-11-23T22:21:22.001532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = df_train[featuresCols]\ndf_test = df_test[featuresColsTest]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:21:22.800092Z","iopub.execute_input":"2024-11-23T22:21:22.800515Z","iopub.status.idle":"2024-11-23T22:21:22.813856Z","shell.execute_reply.started":"2024-11-23T22:21:22.800479Z","shell.execute_reply":"2024-11-23T22:21:22.812498Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = df_train.dropna(subset='sii') # drop rows where sii is NaN\ndf_test  = df_test.drop('id',axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:21:23.489731Z","iopub.execute_input":"2024-11-23T22:21:23.490926Z","iopub.status.idle":"2024-11-23T22:21:23.505823Z","shell.execute_reply.started":"2024-11-23T22:21:23.490874Z","shell.execute_reply":"2024-11-23T22:21:23.504488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(len(featuresCols))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:21:24.213738Z","iopub.execute_input":"2024-11-23T22:21:24.214159Z","iopub.status.idle":"2024-11-23T22:21:24.219465Z","shell.execute_reply.started":"2024-11-23T22:21:24.214125Z","shell.execute_reply":"2024-11-23T22:21:24.218286Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.shape, df_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:21:25.287337Z","iopub.execute_input":"2024-11-23T22:21:25.287702Z","iopub.status.idle":"2024-11-23T22:21:25.296025Z","shell.execute_reply.started":"2024-11-23T22:21:25.287668Z","shell.execute_reply":"2024-11-23T22:21:25.294358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:21:26.638935Z","iopub.execute_input":"2024-11-23T22:21:26.639400Z","iopub.status.idle":"2024-11-23T22:21:26.718451Z","shell.execute_reply.started":"2024-11-23T22:21:26.639361Z","shell.execute_reply":"2024-11-23T22:21:26.717165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:21:28.014951Z","iopub.execute_input":"2024-11-23T22:21:28.015428Z","iopub.status.idle":"2024-11-23T22:21:28.095592Z","shell.execute_reply.started":"2024-11-23T22:21:28.015389Z","shell.execute_reply":"2024-11-23T22:21:28.093975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.columns = df_train.columns.str.replace(\"<\", \"\")\ndf_train.columns = df_train.columns.str.replace(\">\", \"\")\ndf_train.columns = df_train.columns.str.replace(\"0\", \"\")\n\ndf_test.columns = df_test.columns.str.replace(\"<\", \"\")\ndf_test.columns = df_test.columns.str.replace(\">\", \"\")\ndf_test.columns = df_test.columns.str.replace(\"0\", \"\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:21:29.370975Z","iopub.execute_input":"2024-11-23T22:21:29.371455Z","iopub.status.idle":"2024-11-23T22:21:29.384028Z","shell.execute_reply.started":"2024-11-23T22:21:29.371412Z","shell.execute_reply":"2024-11-23T22:21:29.382380Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# helper functions\n\n# replace NaN with 'Missing' and cast as 'category' type\ndef update(df):\n    global cat_c\n    for c in cat_c:\n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category') # category datatype\n    return df\n\n\ndf_train = update(df_train)\ndf_test = update(df_test)\n\n# mapping values for categorical data\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\n# map numerical values for categorical cols\nfor col in cat_c:\n    mapping = create_mapping(col, df_train)\n    mappingTe = create_mapping(col, df_test)\n    df_train[col] = df_train[col].replace(mapping).astype(int)\n    df_test[col] = df_test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:21:30.463389Z","iopub.execute_input":"2024-11-23T22:21:30.464059Z","iopub.status.idle":"2024-11-23T22:21:30.552219Z","shell.execute_reply.started":"2024-11-23T22:21:30.464005Z","shell.execute_reply":"2024-11-23T22:21:30.551014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:21:36.563400Z","iopub.execute_input":"2024-11-23T22:21:36.563886Z","iopub.status.idle":"2024-11-23T22:21:36.643240Z","shell.execute_reply.started":"2024-11-23T22:21:36.563844Z","shell.execute_reply":"2024-11-23T22:21:36.642121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 42","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:21:37.741060Z","iopub.execute_input":"2024-11-23T22:21:37.741457Z","iopub.status.idle":"2024-11-23T22:21:37.747115Z","shell.execute_reply.started":"2024-11-23T22:21:37.741424Z","shell.execute_reply":"2024-11-23T22:21:37.745767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Training with voting model and stratified k-fold CV\n\ndef TrainML(model_class, test_data):\n    X = df_train.drop(['sii'], axis=1)\n    y = df_train['sii']\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\n    train_S = []\n    test_S = []\n\n    oof_non_rounded = np.zeros(len(y), dtype=float)\n    oof_rounded = np.zeros(len(y), dtype=int)\n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int)) # try without rounded\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n\n\n        test_preds[:, fold] = model.predict(test_data)\n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded),\n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n\n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n    submission = pd.DataFrame({\n        'id': df_sample_b['id'],\n        'sii': tpTuned\n    })\n\n    return submission\n\nn_splits = 2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:21:38.582994Z","iopub.execute_input":"2024-11-23T22:21:38.583469Z","iopub.status.idle":"2024-11-23T22:21:38.596707Z","shell.execute_reply.started":"2024-11-23T22:21:38.583431Z","shell.execute_reply":"2024-11-23T22:21:38.595400Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Hyperparameter optimization","metadata":{}},{"cell_type":"code","source":"df_train.drop(['id'], axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:21:40.079929Z","iopub.execute_input":"2024-11-23T22:21:40.080328Z","iopub.status.idle":"2024-11-23T22:21:40.089194Z","shell.execute_reply.started":"2024-11-23T22:21:40.080293Z","shell.execute_reply":"2024-11-23T22:21:40.087572Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = df_train.drop(['sii'], axis=1)\ny = df_train['sii']\n\nx_train, x_test, y_train, y_test = train_test_split(X, y, test_size=0.3, shuffle=True, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:21:40.606030Z","iopub.execute_input":"2024-11-23T22:21:40.606407Z","iopub.status.idle":"2024-11-23T22:21:40.621600Z","shell.execute_reply.started":"2024-11-23T22:21:40.606377Z","shell.execute_reply":"2024-11-23T22:21:40.620400Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def objective(trial, model):\n\n    if model == 'xgb':\n        params = {\n            'objective': 'reg:squarederror',\n\n            'max_depth': trial.suggest_int('max_depth', 1, 10),\n\n            'learning_rate': trial.suggest_float('learning_rate', 0.001, 1.0, log=True),\n\n            'n_estimators': trial.suggest_int('n_estimators', 50, 500),\n\n            'min_child_weight': trial.suggest_int('min_child_weight', 5, 36),\n\n            #'gamma': trial.suggest_loguniform('gamma', 1e-8, 1.0),\n\n            'subsample': trial.suggest_float('subsample', 0.01, 1.0, log=True),\n\n            'colsample_bytree': trial.suggest_float('colsample_bytree', 0.01, 1.0, log=True),\n\n            'reg_alpha': trial.suggest_float('reg_alpha', 1e-8, 1.0, log=True),\n\n            'reg_lambda': trial.suggest_float('reg_lambda', 1e-8, 1.0, log=True),\n\n            'tree_method': 'exact',\n\n            'random_state': SEED,\n\n        }\n\n\n\n        optuna_model = XGBRegressor(**params)\n\n        optuna_model.fit(x_train, y_train)\n\n        # Make predictions\n\n        y_pred = optuna_model.predict(x_test)\n\n\n\n    elif model == 'catboost':\n\n\n\n        params = {\n\n        'learning_rate': trial.suggest_float('learning_rate', 0.001, 1.0, log=True),\n\n        'depth': trial.suggest_int('depth', 1, 10),\n\n        'random_seed': SEED,\n\n        'l2_leaf_reg': trial.suggest_int('l2_leaf_reg', 1, 50),\n\n        }\n\n        optuna_model = CatBoostRegressor(**params)\n\n        optuna_model.fit(x_train, y_train)\n\n        # Make predictions\n\n        y_pred = optuna_model.predict(x_test)\n\n\n\n    elif model == 'lightgbm':\n\n\n\n        params = {\n\n            'max_depth': trial.suggest_int('max_depth', 1, 30),\n\n            'learning_rate': trial.suggest_float('learning_rate', 0.001, 1.0, log=True),\n\n            'num_leaves': trial.suggest_int('num_leaves', 2, 500),\n\n            'min_data_in_leaf': trial.suggest_int('min_data_in_leaf', 2, 100),\n\n            'feature_fraction': trial.suggest_uniform('feature_fraction', 0.1, 1.0),\n\n            'bagging_fraction': trial.suggest_uniform('bagging_fraction', 0.1, 1.0),\n\n            'bagging_freq': trial.suggest_int('bagging_freq', 2, 20),\n\n            'lambda_l1': trial.suggest_float('lambda_l1', 1e-8, 20, log=True),\n\n            'lambda_l2': trial.suggest_float('lambda_l2', 1e-8, 1.0, log=True),\n\n            'random_seed': 42,\n\n        }\n\n\n\n        optuna_model = LGBMRegressor(**params)\n\n        optuna_model.fit(x_train, y_train)\n\n        # Make predictions\n\n        y_pred = optuna_model.predict(x_test)\n\n\n\n    mse = mean_squared_error(y_test, y_pred)\n\n\n\n    trial.set_user_attr('mse', mse)\n\n    print(f\"MSE: {mse}\")\n\n\n\n    return mse\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:22:00.559563Z","iopub.execute_input":"2024-11-23T22:22:00.560083Z","iopub.status.idle":"2024-11-23T22:22:00.574880Z","shell.execute_reply.started":"2024-11-23T22:22:00.560044Z","shell.execute_reply":"2024-11-23T22:22:00.573440Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#study = optuna.create_study(**{'study_name': f'lstm XGB', 'direction':'minimize'})\n\n#study.optimize(lambda trial: objective(trial, 'xgb'), show_progress_bar=True, n_trials=100)\n\n\n\n#print(f\"\\n\\n{study.best_params}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:22:03.764605Z","iopub.execute_input":"2024-11-23T22:22:03.765741Z","iopub.status.idle":"2024-11-23T22:22:03.770252Z","shell.execute_reply.started":"2024-11-23T22:22:03.765688Z","shell.execute_reply":"2024-11-23T22:22:03.769147Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#xgb_params = study.best_params\n\n#xgb_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:22:04.395921Z","iopub.execute_input":"2024-11-23T22:22:04.396732Z","iopub.status.idle":"2024-11-23T22:22:04.400980Z","shell.execute_reply.started":"2024-11-23T22:22:04.396688Z","shell.execute_reply":"2024-11-23T22:22:04.399754Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgb_params = {'max_depth': 8,\n\n 'learning_rate': 0.021151031978533925,\n\n 'n_estimators': 436,\n\n 'min_child_weight': 9,\n\n 'subsample': 0.29554715202815457,\n\n 'colsample_bytree': 0.893606209450445,\n\n 'reg_alpha': 1.3292473386838738e-06,\n\n 'reg_lambda': 0.3795131132153967}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:22:05.376511Z","iopub.execute_input":"2024-11-23T22:22:05.376928Z","iopub.status.idle":"2024-11-23T22:22:05.383446Z","shell.execute_reply.started":"2024-11-23T22:22:05.376884Z","shell.execute_reply":"2024-11-23T22:22:05.382105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#study = optuna.create_study(**{'study_name': f'lstm catboost', 'direction':'minimize'})\n\n#study.optimize(lambda trial: objective(trial, 'catboost'), show_progress_bar=True, n_trials=100)\n\n    \n\n#print(f\"\\n\\n{study.best_params}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:22:06.741457Z","iopub.execute_input":"2024-11-23T22:22:06.741875Z","iopub.status.idle":"2024-11-23T22:22:06.747781Z","shell.execute_reply.started":"2024-11-23T22:22:06.741837Z","shell.execute_reply":"2024-11-23T22:22:06.746330Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#catboost_params = study.best_params\n\n#catboost_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:22:07.420565Z","iopub.execute_input":"2024-11-23T22:22:07.421061Z","iopub.status.idle":"2024-11-23T22:22:07.427086Z","shell.execute_reply.started":"2024-11-23T22:22:07.421025Z","shell.execute_reply":"2024-11-23T22:22:07.425623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"catboost_params = {'learning_rate': 0.11337224383012184, 'depth': 7, 'l2_leaf_reg': 30}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:22:08.078218Z","iopub.execute_input":"2024-11-23T22:22:08.078729Z","iopub.status.idle":"2024-11-23T22:22:08.085451Z","shell.execute_reply.started":"2024-11-23T22:22:08.078687Z","shell.execute_reply":"2024-11-23T22:22:08.084138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#study = optuna.create_study(**{'study_name': f'lstm lightgbm', 'direction':'minimize'})\n\n#study.optimize(lambda trial: objective(trial, 'lightgbm'), show_progress_bar=True, n_trials=100)\n\n\n\n#print(f\"\\n\\n{study.best_params}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:22:08.957713Z","iopub.execute_input":"2024-11-23T22:22:08.958236Z","iopub.status.idle":"2024-11-23T22:22:08.964189Z","shell.execute_reply.started":"2024-11-23T22:22:08.958196Z","shell.execute_reply":"2024-11-23T22:22:08.962605Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#lightgbm_params = study.best_params\n\n#lightgbm_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:22:09.562893Z","iopub.execute_input":"2024-11-23T22:22:09.563304Z","iopub.status.idle":"2024-11-23T22:22:09.568881Z","shell.execute_reply.started":"2024-11-23T22:22:09.563270Z","shell.execute_reply":"2024-11-23T22:22:09.567388Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lightgbm_params = {'max_depth': 21,\n\n    'learning_rate': 0.04924679127532934,\n\n    'num_leaves': 326,\n\n    'min_data_in_leaf': 37,\n\n    'feature_fraction': 0.5556313350626378,\n\n    'bagging_fraction': 0.8747825084196182,\n\n    'bagging_freq': 11,\n\n    'lambda_l1': 0.050541336388124455,\n\n    'lambda_l2': 0.010140672908589561}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:22:10.421115Z","iopub.execute_input":"2024-11-23T22:22:10.421506Z","iopub.status.idle":"2024-11-23T22:22:10.427455Z","shell.execute_reply.started":"2024-11-23T22:22:10.421473Z","shell.execute_reply":"2024-11-23T22:22:10.426202Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create model instances\n\nLight = LGBMRegressor(**lightgbm_params, random_state=SEED, verbose=-1, n_estimators=300)\n\nXGB_Model = XGBRegressor(**xgb_params)\n\nCatBoost_Model = CatBoostRegressor(**catboost_params)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:22:11.513308Z","iopub.execute_input":"2024-11-23T22:22:11.514568Z","iopub.status.idle":"2024-11-23T22:22:11.525414Z","shell.execute_reply.started":"2024-11-23T22:22:11.514469Z","shell.execute_reply":"2024-11-23T22:22:11.523777Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Combine models using Voting Regressor\n\nvoting_model = VotingRegressor(estimators=[\n\n    ('lightgbm', Light),\n\n    ('xgboost', XGB_Model),\n\n    ('catboost', CatBoost_Model)\n\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:22:12.176686Z","iopub.execute_input":"2024-11-23T22:22:12.177132Z","iopub.status.idle":"2024-11-23T22:22:12.183001Z","shell.execute_reply.started":"2024-11-23T22:22:12.177098Z","shell.execute_reply":"2024-11-23T22:22:12.181653Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.shape, df_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:22:12.959623Z","iopub.execute_input":"2024-11-23T22:22:12.961230Z","iopub.status.idle":"2024-11-23T22:22:12.969365Z","shell.execute_reply.started":"2024-11-23T22:22:12.961166Z","shell.execute_reply":"2024-11-23T22:22:12.968141Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.sample()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:22:14.053557Z","iopub.execute_input":"2024-11-23T22:22:14.054018Z","iopub.status.idle":"2024-11-23T22:22:14.117993Z","shell.execute_reply.started":"2024-11-23T22:22:14.053982Z","shell.execute_reply":"2024-11-23T22:22:14.116348Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.sample(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:22:15.105273Z","iopub.execute_input":"2024-11-23T22:22:15.105829Z","iopub.status.idle":"2024-11-23T22:22:15.200360Z","shell.execute_reply.started":"2024-11-23T22:22:15.105772Z","shell.execute_reply":"2024-11-23T22:22:15.198645Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:22:17.081551Z","iopub.execute_input":"2024-11-23T22:22:17.083098Z","iopub.status.idle":"2024-11-23T22:22:17.092831Z","shell.execute_reply.started":"2024-11-23T22:22:17.083050Z","shell.execute_reply":"2024-11-23T22:22:17.091180Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#df_test.drop(['time_of_day'], axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:22:18.055357Z","iopub.execute_input":"2024-11-23T22:22:18.055845Z","iopub.status.idle":"2024-11-23T22:22:18.061127Z","shell.execute_reply.started":"2024-11-23T22:22:18.055775Z","shell.execute_reply":"2024-11-23T22:22:18.059701Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# run trainml\n\nSubmission1 = TrainML(voting_model, df_test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:22:19.384619Z","iopub.execute_input":"2024-11-23T22:22:19.385087Z","iopub.status.idle":"2024-11-23T22:23:00.959688Z","shell.execute_reply.started":"2024-11-23T22:22:19.385050Z","shell.execute_reply":"2024-11-23T22:23:00.958385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save submission\nSubmission1.to_csv('submission_b.csv', index=False)\nprint(Submission1['sii'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T22:23:50.569910Z","iopub.execute_input":"2024-11-23T22:23:50.571121Z","iopub.status.idle":"2024-11-23T22:23:50.582027Z","shell.execute_reply.started":"2024-11-23T22:23:50.571057Z","shell.execute_reply":"2024-11-23T22:23:50.580827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}