{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":10235545,"sourceType":"datasetVersion","datasetId":6329094}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#  Child Mind Institute - problematic internet use ","metadata":{}},{"cell_type":"markdown","source":"# Import libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns\nimport plotly.subplots as sp\nimport plotly.express as px\nfrom concurrent.futures import ThreadPoolExecutor\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nfrom IPython.display import display\nimport seaborn as sns\nimport warnings\nimport math\n\nfrom plotly.subplots import make_subplots\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nimport xgboost as xgb\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report\nfrom sklearn.preprocessing import  LabelEncoder ,StandardScaler, OneHotEncoder,RobustScaler\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.metrics import confusion_matrix , accuracy_score , recall_score , f1_score , classification_report,precision_score,roc_curve, auc\nfrom collections import Counter\nfrom sklearn.svm import SVC","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-12-19T22:09:54.648840Z","iopub.execute_input":"2024-12-19T22:09:54.649159Z","iopub.status.idle":"2024-12-19T22:09:54.655475Z","shell.execute_reply.started":"2024-12-19T22:09:54.649137Z","shell.execute_reply":"2024-12-19T22:09:54.654455Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"! pip install /kaggle/input/tabnet/pytorch_tabnet-4.1.0-py3-none-any.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:09:54.656684Z","iopub.execute_input":"2024-12-19T22:09:54.657258Z","iopub.status.idle":"2024-12-19T22:09:57.983062Z","shell.execute_reply.started":"2024-12-19T22:09:54.657210Z","shell.execute_reply":"2024-12-19T22:09:57.981862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.base import clone, BaseEstimator, RegressorMixin\nfrom sklearn.ensemble import RandomForestClassifier, RandomForestRegressor\nfrom sklearn.ensemble import StackingRegressor\nfrom sklearn.linear_model import Ridge\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor, HistGradientBoostingRegressor, ExtraTreesRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.model_selection import GridSearchCV\n\nfrom pytorch_tabnet.tab_model import TabNetRegressor\nfrom pytorch_tabnet.callbacks import Callback\n\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\nimport pytorch_tabnet\nimport os\n\nimport random\ndef seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True\nseed_everything(2024)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:09:57.985126Z","iopub.execute_input":"2024-12-19T22:09:57.985478Z","iopub.status.idle":"2024-12-19T22:09:57.996088Z","shell.execute_reply.started":"2024-12-19T22:09:57.985441Z","shell.execute_reply":"2024-12-19T22:09:57.995328Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data exploration and preprocessing\n","metadata":{}},{"cell_type":"markdown","source":"## Data Exploration","metadata":{}},{"cell_type":"code","source":"train_df=pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ndata_dict = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')","metadata":{"execution":{"iopub.status.busy":"2024-12-19T22:09:57.997201Z","iopub.execute_input":"2024-12-19T22:09:57.997516Z","iopub.status.idle":"2024-12-19T22:09:58.064269Z","shell.execute_reply.started":"2024-12-19T22:09:57.997493Z","shell.execute_reply":"2024-12-19T22:09:58.063565Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* Reads the Parquet file and converts it into DataFrame format.Remove the unnecessary 'step' column and calculate statistical information for the remaining columns.\n \n* Processes all files in the specified directory and extracts statistical information for each file.\n \n* Trains the model to minimize the difference between the reconstructed data and the original data through the encoder and decoder.\n \n* Trains a model using Huber Loss as the loss function and uses the trained encoder to transform the input data into a low-dimensional latent representation.tation.","metadata":{}},{"cell_type":"code","source":"import os\nfrom  concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\n# Reads the Parquet file and converts it into DataFrame format \ndef process_file(filename, dirname):\n    data = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    data.drop('step', axis=1, inplace=True)\n    return data.describe().values.reshape(-1), filename.split('=')[1]\n#Extracts statistical information for each file\ndef load_time_series(dirname):\n    ids = os.listdir(dirname)\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    stats, indexes = zip(*results)\n    data = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    data['id'] = indexes\n    return data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:09:58.065170Z","iopub.execute_input":"2024-12-19T22:09:58.065528Z","iopub.status.idle":"2024-12-19T22:09:58.071705Z","shell.execute_reply.started":"2024-12-19T22:09:58.065492Z","shell.execute_reply":"2024-12-19T22:09:58.070773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#minimize difference between the reconstructed data and original data\nclass AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim*3),\n            nn.LeakyReLU(0.2),\n            nn.Linear(encoding_dim*3, encoding_dim*2),\n            nn.LeakyReLU(0.2),\n            nn.Linear(encoding_dim*2, encoding_dim),\n            nn.LeakyReLU(0.2)\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim*2),\n            nn.LeakyReLU(0.2),\n            nn.Linear(input_dim*2, input_dim*3),\n            nn.LeakyReLU(0.2),\n            nn.Linear(input_dim*3, input_dim),\n            nn.Sigmoid()\n        )\n\n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:09:58.072479Z","iopub.execute_input":"2024-12-19T22:09:58.072766Z","iopub.status.idle":"2024-12-19T22:09:58.092826Z","shell.execute_reply.started":"2024-12-19T22:09:58.072744Z","shell.execute_reply":"2024-12-19T22:09:58.092001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n\n    data_tensor = torch.FloatTensor(df_scaled)\n\n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n\n    criterion = F.smooth_l1_loss\n    optimizer = optim.Adam(autoencoder.parameters())\n\n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n\n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n\n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n\n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n\n    return df_encoded","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:09:58.093649Z","iopub.execute_input":"2024-12-19T22:09:58.093886Z","iopub.status.idle":"2024-12-19T22:09:58.110641Z","shell.execute_reply.started":"2024-12-19T22:09:58.093867Z","shell.execute_reply":"2024-12-19T22:09:58.109862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineering(df):\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1) \n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:09:58.111584Z","iopub.execute_input":"2024-12-19T22:09:58.111873Z","iopub.status.idle":"2024-12-19T22:09:58.133840Z","shell.execute_reply.started":"2024-12-19T22:09:58.111844Z","shell.execute_reply":"2024-12-19T22:09:58.133229Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:09:58.136200Z","iopub.execute_input":"2024-12-19T22:09:58.136475Z","iopub.status.idle":"2024-12-19T22:11:06.157196Z","shell.execute_reply.started":"2024-12-19T22:09:58.136453Z","shell.execute_reply":"2024-12-19T22:11:06.156461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\n\ntrain_ts_encoded = perform_autoencoder(df_train, encoding_dim=60, epochs=100, batch_size=32)\ntest_ts_encoded = perform_autoencoder(df_test, encoding_dim=60, epochs=100, batch_size=32)\n\ntime_series_cols = train_ts_encoded.columns.tolist()\ntrain_ts_encoded[\"id\"]=train_ts[\"id\"]\ntest_ts_encoded['id']=test_ts[\"id\"]\n\ntrain = pd.merge(train_df, train_ts_encoded, how=\"left\", on='id')\ntest = pd.merge(test_df, test_ts_encoded, how=\"left\", on='id')\n\nimputer = KNNImputer(n_neighbors=5)\nnumeric_cols = train.select_dtypes(include=['float64', 'int64']).columns\nimputed_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\n        \ntrain = train_imputed\n\ntrain = feature_engineering(train)\ntrain = train.dropna(thresh=10, axis=0)\ntest = feature_engineering(test)\n\ntrain = train.drop('id', axis=1)\ntest  = test .drop('id', axis=1)   \n\n\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW']\n\nfeaturesCols += time_series_cols\ntest = test[featuresCols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:11:06.158615Z","iopub.execute_input":"2024-12-19T22:11:06.158836Z","iopub.status.idle":"2024-12-19T22:11:22.439211Z","shell.execute_reply.started":"2024-12-19T22:11:06.158817Z","shell.execute_reply":"2024-12-19T22:11:22.438264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if np.any(np.isinf(train)):\n   # print(\"exist inf value\")\n\n   # inf_rows, inf_cols = np.where(np.isinf(train))\n\n   # for row, col in zip(inf_rows, inf_cols):\n   #     print(f\"columns {row}, row {train.columns[col]}: {train.iloc[row, col]}\")\n   train = train.replace([np.inf, -np.inf], np.nan)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:11:22.440160Z","iopub.execute_input":"2024-12-19T22:11:22.440498Z","iopub.status.idle":"2024-12-19T22:11:22.448147Z","shell.execute_reply.started":"2024-12-19T22:11:22.440466Z","shell.execute_reply":"2024-12-19T22:11:22.447242Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\n#optimal_thresholds = [0.5, 1.5, 2.5]\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:11:22.449044Z","iopub.execute_input":"2024-12-19T22:11:22.449333Z","iopub.status.idle":"2024-12-19T22:11:22.464764Z","shell.execute_reply.started":"2024-12-19T22:11:22.449302Z","shell.execute_reply":"2024-12-19T22:11:22.464093Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Preprocessing","metadata":{}},{"cell_type":"markdown","source":"# Train Model","metadata":{}},{"cell_type":"markdown","source":"We will use ensemble learning to have a chance to the highest score as possible","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import VotingClassifier, StackingClassifier, RandomForestClassifier\nfrom xgboost import XGBClassifier\nfrom lightgbm import LGBMClassifier \nfrom catboost import CatBoostClassifier\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score, make_scorer\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.ensemble import RandomForestClassifier\nfrom imblearn.over_sampling import SMOTE\nimport optuna\n\n\nfeature_col = train.drop(['sii'], axis=1).columns\n\nall_importances = pd.DataFrame({'feature': feature_col})\n       ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:11:22.465619Z","iopub.execute_input":"2024-12-19T22:11:22.465864Z","iopub.status.idle":"2024-12-19T22:11:22.493005Z","shell.execute_reply.started":"2024-12-19T22:11:22.465843Z","shell.execute_reply":"2024-12-19T22:11:22.492270Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def TrainML_old(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\n    train_S = []\n    test_S = []\n\n    oof_non_rounded = np.zeros(len(y), dtype=float)\n    oof_rounded = np.zeros(len(y), dtype=int)\n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n\n        test_preds[:, fold] = model.predict(test_data)\n\n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n            \n\n    \n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded),\n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n\n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n    submission_df = pd.DataFrame({\n        'id': submission['id'],\n        'sii': tpTuned\n    })\n\n    return submission_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:11:22.493956Z","iopub.execute_input":"2024-12-19T22:11:22.494295Z","iopub.status.idle":"2024-12-19T22:11:22.511367Z","shell.execute_reply.started":"2024-12-19T22:11:22.494271Z","shell.execute_reply":"2024-12-19T22:11:22.510555Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n        \n    feature_names = X.columns\n\n    \n    lgb_importances_list = []\n    xgb_importances_list = []\n    cat_importances_list = []\n    tabnet_importances_list = []\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n       \n        named_estimators = model.named_estimators_\n        \n        # LightGBM\n        if 'lightgbm' in named_estimators and hasattr(named_estimators['lightgbm'], 'feature_importances_'):\n            lgb_importances_list.append(named_estimators['lightgbm'].feature_importances_)\n\n        # XGBoost\n        if 'xgboost' in named_estimators and hasattr(named_estimators['xgboost'], 'feature_importances_'):\n            xgb_importances_list.append(named_estimators['xgboost'].feature_importances_)\n\n        # CatBoost\n        if 'catboost' in named_estimators and hasattr(named_estimators['catboost'], 'get_feature_importance'):\n            cat_importances_list.append(named_estimators['catboost'].get_feature_importance())\n\n        # TabNet\n        if 'tabnet' in named_estimators and hasattr(named_estimators['tabnet'], 'feature_importances_'):\n            tabnet_importances_list.append(named_estimators['tabnet'].feature_importances_)\n\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n   \n    def mean_importances(importances_list):\n        if len(importances_list) > 0:\n            return np.mean(importances_list, axis=0)\n        else:\n            return None\n\n    lgb_mean = mean_importances(lgb_importances_list)\n    xgb_mean = mean_importances(xgb_importances_list)\n    cat_mean = mean_importances(cat_importances_list)\n    tab_mean = mean_importances(tabnet_importances_list)\n\n    def normalize_importances(importance_array):\n        if importance_array is not None:\n            return importance_array / importance_array.sum()\n        else:\n            return None\n\n    lgb_mean_normalized = normalize_importances(lgb_mean)\n    xgb_mean_normalized = normalize_importances(xgb_mean)\n    cat_mean_normalized = normalize_importances(cat_mean)\n    tab_mean_normalized = normalize_importances(tab_mean)\n   \n\n    return submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:11:22.512071Z","iopub.execute_input":"2024-12-19T22:11:22.512309Z","iopub.status.idle":"2024-12-19T22:11:22.529042Z","shell.execute_reply.started":"2024-12-19T22:11:22.512288Z","shell.execute_reply":"2024-12-19T22:11:22.528430Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_vals1 = []\nY_vals1 = []\nY_preds1 = []\ndef TrainML_old(model_class, test_data):\n    \n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        X_vals1.append(X_val)\n        Y_vals1.append(y_val)\n        Y_preds1.append(y_val_pred)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n    return submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:11:22.529848Z","iopub.execute_input":"2024-12-19T22:11:22.530078Z","iopub.status.idle":"2024-12-19T22:11:22.552858Z","shell.execute_reply.started":"2024-12-19T22:11:22.530046Z","shell.execute_reply":"2024-12-19T22:11:22.552287Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Parameters setting","metadata":{}},{"cell_type":"markdown","source":"Model1","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import Ridge\nfrom sklearn.svm import SVR","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:11:22.553599Z","iopub.execute_input":"2024-12-19T22:11:22.553892Z","iopub.status.idle":"2024-12-19T22:11:22.575455Z","shell.execute_reply.started":"2024-12-19T22:11:22.553861Z","shell.execute_reply":"2024-12-19T22:11:22.574814Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"n_splits = 5\nSEED = 42\n\nLGBM_Params = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01,  # Increased from 2.68e-06\n    'device': 'cpu',\n}\n\n\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED,\n    'tree_method': 'gpu_hist'\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10,  # Increase this value\n    'task_type': 'GPU',\n}\n\nRF_Params = {\n    'n_estimators': 200,\n    'max_depth': 6,\n    'max_features': 0.8,\n    'min_samples_split': 2,\n    'min_samples_leaf': 1,\n    'bootstrap': True,\n    'random_state': SEED\n}\n\nExtraTrees_Params = {\n    'n_estimators': 100,\n    'criterion': 'squared_error',\n    'max_depth': None,\n    'min_samples_split': 2,\n    'min_samples_leaf': 1,\n    'max_features': 'auto',\n    'bootstrap': True,\n    'random_state': SEED,\n    'n_jobs': -1\n}\n\nHistGB_Params = {\n    'max_iter': 200, \n    'max_depth': 6,  \n    'learning_rate': 0.5, \n    'loss': 'squared_error', \n    'max_leaf_nodes': None,  \n    'random_state': SEED\n}\n\nRidge_Params = {\n    'alpha': 1.0, \n    'solver': 'auto', \n    'random_state': SEED \n}\n\nSVR_Params = {\n    'C': 1.0, \n    'kernel': 'rbf', \n    'epsilon': 0.1,\n    'degree': 3,  \n}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:11:22.576201Z","iopub.execute_input":"2024-12-19T22:11:22.576422Z","iopub.status.idle":"2024-12-19T22:11:22.592401Z","shell.execute_reply.started":"2024-12-19T22:11:22.576404Z","shell.execute_reply":"2024-12-19T22:11:22.591778Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Tabnet model\nclass TabNetWrapper(BaseEstimator, RegressorMixin):\n    def __init__(self, **kwargs):\n        self.model = TabNetRegressor(**kwargs)\n        self.kwargs = kwargs\n        self.imputer = KNNImputer(n_neighbors=5)\n        #self.imputer = SimpleImputer(strategy='median')\n        self.best_model_path = 'best_tabnet_model.pt'\n\n    def fit(self, X, y):\n        X_imputed = self.imputer.fit_transform(X)\n\n        if hasattr(y, 'values'):\n            y = y.values\n\n        X_train, X_valid, y_train, y_valid = train_test_split(\n            X_imputed,\n            y,\n            test_size=0.2,\n            random_state=42\n        )\n\n        # Train TabNet model\n        history = self.model.fit(\n            X_train=X_train,\n            y_train=y_train.reshape(-1, 1),\n            eval_set=[(X_valid, y_valid.reshape(-1, 1))],\n            eval_name=['valid'],\n            eval_metric=['mse', 'mae', 'rmse'],\n            max_epochs=500,\n            patience=50,\n            batch_size=1024,\n            virtual_batch_size=128,\n            num_workers=0,\n            drop_last=False,\n            callbacks=[\n                TabNetPretrainedModelCheckpoint(\n                    filepath=self.best_model_path,\n                    monitor='valid_mse',\n                    mode='min',\n                    save_best_only=True,\n                    verbose=True\n                )\n            ]\n        )\n\n        # Load the best model\n        if os.path.exists(self.best_model_path):\n            self.model.load_model(self.best_model_path)\n            os.remove(self.best_model_path)  # Remove temporary file\n\n        return self\n\n    def predict(self, X):\n        X_imputed = self.imputer.transform(X)\n        return self.model.predict(X_imputed).flatten()\n\n    def __deepcopy__(self, memo):\n        cls = self.__class__\n        result = cls.__new__(cls)\n        memo[id(self)] = result\n        for k, v in self.__dict__.items():\n            setattr(result, k, deepcopy(v, memo))\n        return result\n\n\nTabNet_Params = {\n    'n_d': 64,\n    'n_a': 64,\n    'n_steps': 5,\n    'gamma': 1.5,\n    'n_independent': 2,\n    'n_shared': 2,\n    'lambda_sparse': 1e-4,\n    'optimizer_fn': torch.optim.Adam,\n    'optimizer_params': dict(lr=2e-2, weight_decay=1e-5),\n    'mask_type': 'entmax',\n    'scheduler_params': dict(mode=\"min\", patience=10, min_lr=1e-5, factor=0.5),\n    'scheduler_fn': torch.optim.lr_scheduler.ReduceLROnPlateau,\n    'verbose': 1,\n    'device_name': 'cuda' if torch.cuda.is_available() else 'cpu'\n}\n\nclass TabNetPretrainedModelCheckpoint(Callback):\n    def __init__(self, filepath, monitor='val_loss', mode='min',\n                 save_best_only=True, verbose=1):\n        super().__init__()\n        self.filepath = filepath\n        self.monitor = monitor\n        self.mode = mode\n        self.save_best_only = save_best_only\n        self.verbose = verbose\n        self.best = float('inf') if mode == 'min' else -float('inf')\n\n    def on_train_begin(self, logs=None):\n        self.model = self.trainer\n\n    def on_epoch_end(self, epoch, logs=None):\n        logs = logs or {}\n        current = logs.get(self.monitor)\n        if current is None:\n            return\n\n        if (self.mode == 'min' and current < self.best) or \\\n           (self.mode == 'max' and current > self.best):\n            if self.verbose:\n                print(f'\\nEpoch {epoch}: {self.monitor} improved from {self.best:.4f} to {current:.4f}')\n            self.best = current\n            if self.save_best_only:\n                self.model.save_model(self.filepath)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:11:22.593147Z","iopub.execute_input":"2024-12-19T22:11:22.593459Z","iopub.status.idle":"2024-12-19T22:11:22.614245Z","shell.execute_reply.started":"2024-12-19T22:11:22.593421Z","shell.execute_reply":"2024-12-19T22:11:22.613545Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create model instances\nLight = LGBMRegressor(**LGBM_Params, random_state=SEED, verbose=-1, n_estimators=300)\nRF_Model = RandomForestRegressor(**RF_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\nXGB_Model = XGBRegressor(**XGB_Params)\nTabNet_Model = TabNetWrapper(**TabNet_Params) \nExtraTrees_Model = ExtraTreesRegressor(**ExtraTrees_Params)\nHistGB_Model = HistGradientBoostingRegressor(**HistGB_Params)\nRidge_Model = Ridge(**Ridge_Params)\nSVR_Model = SVR(**SVR_Params)\n\n\n# Combine models using Voting Regressor\nridge_pipeline = Pipeline([\n    ('imputer', SimpleImputer(strategy='mean')),  \n    ('ridge', Ridge(alpha=1.0))\n])\n\nestimators = [\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    ('tabnet', TabNet_Model)\n]\n\n# VotingRegressor\nvoting_model = VotingRegressor(\n    estimators=estimators,\n    weights=[4.0, 4.0, 5.0, 4.0] \n)\n\n\nSubmission1 = TrainML(voting_model, test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:11:22.615064Z","iopub.execute_input":"2024-12-19T22:11:22.615279Z","iopub.status.idle":"2024-12-19T22:14:45.336255Z","shell.execute_reply.started":"2024-12-19T22:11:22.615260Z","shell.execute_reply":"2024-12-19T22:14:45.335306Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Submission 2","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n        \ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.49, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    thresholds = KappaOPtimizer.x\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, thresholds)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    fold_weights = [1.25, 1.0, 1.0, 1.0, 1.0]\n    tpm = test_preds.dot(fold_weights) / np.sum(fold_weights)\n    tpTuned = threshold_Rounder(tpm, thresholds)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission\n\n# Model parameters for LightGBM\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01  # Increased from 2.68e-06\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'cat_features': cat_c,\n    'verbose': 0,\n    'l2_leaf_reg': 10  # Increase this value\n}\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\nHistGB_Model = HistGradientBoostingRegressor(**HistGB_Params)\nRidge_Model = Ridge(**Ridge_Params)\nSVR_Model = SVR(**SVR_Params)\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model)\n])\n\n# Train the ensemble model\nSubmission2 = TrainML(voting_model, test)\n\n# Save submission\n#Submission2.to_csv('submission.csv', index=False)\nSubmission2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:14:45.337300Z","iopub.execute_input":"2024-12-19T22:14:45.337526Z","iopub.status.idle":"2024-12-19T22:16:46.098349Z","shell.execute_reply.started":"2024-12-19T22:14:45.337506Z","shell.execute_reply":"2024-12-19T22:16:46.097567Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Submission 3","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    thresholds = KappaOPtimizer.x\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, thresholds)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tp_rounded = threshold_Rounder(tpm, thresholds)\n\n    return tp_rounded\n\nimputer = SimpleImputer(strategy='median')\n\nensemble = VotingRegressor(estimators=[\n    ('lgb', Pipeline(steps=[('imputer', imputer), ('regressor', LGBMRegressor(random_state=SEED))])),\n    ('xgb', Pipeline(steps=[('imputer', imputer), ('regressor', XGBRegressor(random_state=SEED))])),\n    ('cat', Pipeline(steps=[('imputer', imputer), ('regressor', CatBoostRegressor(random_state=SEED, silent=True))])),\n    ('rf', Pipeline(steps=[('imputer', imputer), ('regressor', RandomForestRegressor(random_state=SEED))])),\n    ('gb', Pipeline(steps=[('imputer', imputer), ('regressor', GradientBoostingRegressor(random_state=SEED))]))\n])\n\nSubmission3 = TrainML(ensemble, test)\nSubmission3 = pd.DataFrame({\n    'id': sample['id'],\n    'sii': Submission3\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:16:46.099139Z","iopub.execute_input":"2024-12-19T22:16:46.099398Z","iopub.status.idle":"2024-12-19T22:19:55.022905Z","shell.execute_reply.started":"2024-12-19T22:16:46.099369Z","shell.execute_reply":"2024-12-19T22:19:55.022109Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub1 = Submission1\nsub2 = Submission2\nsub3 = Submission3\n\nsub1 = sub1.sort_values(by='id').reset_index(drop=True)\nsub2 = sub2.sort_values(by='id').reset_index(drop=True)\nsub3 = sub3.sort_values(by='id').reset_index(drop=True)\n\ncombined = pd.DataFrame({\n    'id': sub1['id'],\n    'sii_1': sub1['sii'],\n    'sii_2': sub2['sii'],\n    'sii_3': sub3['sii']\n})\n\ndef majority_vote(row):\n    return row.mode()[0]\n\ncombined['final_sii'] = combined[['sii_1', 'sii_2', 'sii_3']].apply(majority_vote, axis=1)\n\nsum_submission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\nsum_submission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:19:55.023904Z","iopub.execute_input":"2024-12-19T22:19:55.024278Z","iopub.status.idle":"2024-12-19T22:19:55.038528Z","shell.execute_reply.started":"2024-12-19T22:19:55.024243Z","shell.execute_reply":"2024-12-19T22:19:55.037760Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sum_submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:19:55.039332Z","iopub.execute_input":"2024-12-19T22:19:55.039601Z","iopub.status.idle":"2024-12-19T22:19:55.059388Z","shell.execute_reply.started":"2024-12-19T22:19:55.039573Z","shell.execute_reply":"2024-12-19T22:19:55.058725Z"}},"outputs":[],"execution_count":null}]}