{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"colab":{"private_outputs":true,"provenance":[],"gpuType":"V28"},"accelerator":"TPU","kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30805,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nfrom tqdm import tqdm\nfrom concurrent.futures import ThreadPoolExecutor\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import StratifiedKFold, train_test_split\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.base import BaseEstimator, RegressorMixin, clone\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.metrics import cohen_kappa_score\nfrom scipy.optimize import minimize\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor","metadata":{"id":"J-GHH9V2D59J","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:38:58.199340Z","iopub.execute_input":"2024-12-11T05:38:58.199711Z","iopub.status.idle":"2024-12-11T05:38:58.206274Z","shell.execute_reply.started":"2024-12-11T05:38:58.199671Z","shell.execute_reply":"2024-12-11T05:38:58.205271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ndata_dict = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')","metadata":{"id":"FToeysW2EvHB","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:38:58.208231Z","iopub.execute_input":"2024-12-11T05:38:58.208525Z","iopub.status.idle":"2024-12-11T05:38:58.309743Z","shell.execute_reply.started":"2024-12-11T05:38:58.208496Z","shell.execute_reply":"2024-12-11T05:38:58.308999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.head()","metadata":{"id":"0Tz1x6NxFNkp","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:38:58.310725Z","iopub.execute_input":"2024-12-11T05:38:58.310974Z","iopub.status.idle":"2024-12-11T05:38:58.353749Z","shell.execute_reply.started":"2024-12-11T05:38:58.310950Z","shell.execute_reply":"2024-12-11T05:38:58.352823Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_data.shape)","metadata":{"id":"CO880HEEFRdo","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:38:58.354989Z","iopub.execute_input":"2024-12-11T05:38:58.355340Z","iopub.status.idle":"2024-12-11T05:38:58.360680Z","shell.execute_reply.started":"2024-12-11T05:38:58.355308Z","shell.execute_reply":"2024-12-11T05:38:58.359678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.info()","metadata":{"id":"-GZlKOx3GOh3","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:38:58.364331Z","iopub.execute_input":"2024-12-11T05:38:58.364677Z","iopub.status.idle":"2024-12-11T05:38:58.395062Z","shell.execute_reply.started":"2024-12-11T05:38:58.364649Z","shell.execute_reply":"2024-12-11T05:38:58.394019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(test_data.shape)","metadata":{"id":"_ItIvEtUFYZA","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:38:58.396239Z","iopub.execute_input":"2024-12-11T05:38:58.396543Z","iopub.status.idle":"2024-12-11T05:38:58.401366Z","shell.execute_reply.started":"2024-12-11T05:38:58.396514Z","shell.execute_reply":"2024-12-11T05:38:58.400290Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data.head()","metadata":{"id":"0WY4-NXGFfn_","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:38:58.402620Z","iopub.execute_input":"2024-12-11T05:38:58.402910Z","iopub.status.idle":"2024-12-11T05:38:58.428795Z","shell.execute_reply.started":"2024-12-11T05:38:58.402882Z","shell.execute_reply":"2024-12-11T05:38:58.427749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data.info()","metadata":{"id":"b1m0WzfbGiuG","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:38:58.430049Z","iopub.execute_input":"2024-12-11T05:38:58.430413Z","iopub.status.idle":"2024-12-11T05:38:58.445845Z","shell.execute_reply.started":"2024-12-11T05:38:58.430380Z","shell.execute_reply":"2024-12-11T05:38:58.444917Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dict.head()","metadata":{"id":"ZOI3ZtAPFrGw","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:38:58.447068Z","iopub.execute_input":"2024-12-11T05:38:58.447360Z","iopub.status.idle":"2024-12-11T05:38:58.463326Z","shell.execute_reply.started":"2024-12-11T05:38:58.447332Z","shell.execute_reply":"2024-12-11T05:38:58.462389Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n\n    stats, indexes = zip(*results)\n\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df","metadata":{"id":"LC7L7aiIO0Ui","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:38:58.464611Z","iopub.execute_input":"2024-12-11T05:38:58.464979Z","iopub.status.idle":"2024-12-11T05:38:58.473495Z","shell.execute_reply.started":"2024-12-11T05:38:58.464949Z","shell.execute_reply":"2024-12-11T05:38:58.472717Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load time-series data\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")","metadata":{"id":"5rRFB-oyF9cf","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:38:58.474787Z","iopub.execute_input":"2024-12-11T05:38:58.475505Z","iopub.status.idle":"2024-12-11T05:40:20.378096Z","shell.execute_reply.started":"2024-12-11T05:38:58.475462Z","shell.execute_reply":"2024-12-11T05:40:20.377090Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineering(df):\n    # Dropping irrelevant features\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1)\n\n    # Creating new features\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    df['BMI_PHR'] = df['Physical-BMI'] * df['Physical-HeartRate']\n    df['SDS_InternetHours'] = df['SDS-SDS_Total_T'] * df['PreInt_EduHx-computerinternet_hoursday']\n\n    return df\n\ntrain = feature_engineering(train_data)\ntest = feature_engineering(test_data)","metadata":{"id":"91DHce5_Iu-3","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:20.379149Z","iopub.execute_input":"2024-12-11T05:40:20.379491Z","iopub.status.idle":"2024-12-11T05:40:20.405763Z","shell.execute_reply.started":"2024-12-11T05:40:20.379462Z","shell.execute_reply":"2024-12-11T05:40:20.404817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim*3),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*3, encoding_dim*2),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*2, encoding_dim),\n            nn.ReLU()\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim*2),\n            nn.ReLU(),\n            nn.Linear(input_dim*2, input_dim*3),\n            nn.ReLU(),\n            nn.Linear(input_dim*3, input_dim),\n            nn.Sigmoid()\n        )\n\n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded\n\ndef perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n\n    data_tensor = torch.FloatTensor(df_scaled)\n    input_dim = data_tensor.shape[1]\n\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(autoencoder.parameters())\n\n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n\n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n\n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n\n    return pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoding_dim)])","metadata":{"id":"vYdZnKwfQuhy","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:20.407214Z","iopub.execute_input":"2024-12-11T05:40:20.407962Z","iopub.status.idle":"2024-12-11T05:40:20.428015Z","shell.execute_reply.started":"2024-12-11T05:40:20.407913Z","shell.execute_reply":"2024-12-11T05:40:20.426946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Dimensionality reduction\ntrain_ts_encoded = perform_autoencoder(train_ts.drop('id', axis=1), encoding_dim=60, epochs=100)\ntest_ts_encoded = perform_autoencoder(test_ts.drop('id', axis=1), encoding_dim=60, epochs=100)","metadata":{"id":"hFZWpcxORJPZ","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:20.432006Z","iopub.execute_input":"2024-12-11T05:40:20.432624Z","iopub.status.idle":"2024-12-11T05:40:33.506199Z","shell.execute_reply.started":"2024-12-11T05:40:20.432591Z","shell.execute_reply":"2024-12-11T05:40:33.505217Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ts_encoded['id'] = train_ts['id']\ntest_ts_encoded['id'] = test_ts['id']","metadata":{"id":"FWy9eO24RaQx","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:33.507776Z","iopub.execute_input":"2024-12-11T05:40:33.508637Z","iopub.status.idle":"2024-12-11T05:40:33.515104Z","shell.execute_reply.started":"2024-12-11T05:40:33.508586Z","shell.execute_reply":"2024-12-11T05:40:33.514208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"time_series_cols = train_ts_encoded.columns.tolist()","metadata":{"id":"NedYynagdsd_","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:33.516128Z","iopub.execute_input":"2024-12-11T05:40:33.516519Z","iopub.status.idle":"2024-12-11T05:40:33.528092Z","shell.execute_reply.started":"2024-12-11T05:40:33.516478Z","shell.execute_reply":"2024-12-11T05:40:33.527094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Merge encoded time series features with main dataset\ntrain = pd.merge(train, train_ts_encoded, how='left', on='id')\ntest = pd.merge(test, test_ts_encoded, how='left', on='id')","metadata":{"id":"yErQXofeRdop","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:33.529459Z","iopub.execute_input":"2024-12-11T05:40:33.529848Z","iopub.status.idle":"2024-12-11T05:40:33.562835Z","shell.execute_reply.started":"2024-12-11T05:40:33.529806Z","shell.execute_reply":"2024-12-11T05:40:33.561734Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"id":"NDK4ZatxTFVi","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:33.564205Z","iopub.execute_input":"2024-12-11T05:40:33.564604Z","iopub.status.idle":"2024-12-11T05:40:33.590873Z","shell.execute_reply.started":"2024-12-11T05:40:33.564563Z","shell.execute_reply":"2024-12-11T05:40:33.589827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer, KNNImputer\nimport numpy as np\n\nimputer = KNNImputer(n_neighbors=5)\nnumeric_cols = train.select_dtypes(include=['int32', 'int64', 'float64', 'int64']).columns\n\n# Replace infinite values with NaN\ntrain[numeric_cols] = train[numeric_cols].replace([np.inf, -np.inf], np.nan)\n\n# Clip excessively large values to a reasonable range\n# You might need to adjust the lower and upper bounds based on your data\ntrain[numeric_cols] = train[numeric_cols].clip(lower=-1e10, upper=1e10)\n\nimputed_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\n\ntrain = train_imputed\n\ntrain = feature_engineering(train)\n# train = train.dropna(thresh=10, axis=0)\ntest = feature_engineering(test)","metadata":{"id":"rud2TpUOblCP","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:33.592012Z","iopub.execute_input":"2024-12-11T05:40:33.592448Z","iopub.status.idle":"2024-12-11T05:40:43.018911Z","shell.execute_reply.started":"2024-12-11T05:40:33.592395Z","shell.execute_reply":"2024-12-11T05:40:43.017984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.drop('id', axis=1)\ntrain.head()","metadata":{"id":"uxjIrrKDctZH","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:43.020612Z","iopub.execute_input":"2024-12-11T05:40:43.021017Z","iopub.status.idle":"2024-12-11T05:40:43.053147Z","shell.execute_reply.started":"2024-12-11T05:40:43.020970Z","shell.execute_reply":"2024-12-11T05:40:43.052110Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"featuresCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW', 'BMI_PHR', 'SDS_InternetHours']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW', 'BMI_PHR', 'SDS_InternetHours']\n\nfeaturesCols += time_series_cols\ntest = test[featuresCols]","metadata":{"id":"CTBJEpWmclow","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:43.054370Z","iopub.execute_input":"2024-12-11T05:40:43.054689Z","iopub.status.idle":"2024-12-11T05:40:43.077202Z","shell.execute_reply.started":"2024-12-11T05:40:43.054660Z","shell.execute_reply":"2024-12-11T05:40:43.076416Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Impute missing values using the mean for numerical features\nnumerical_cols = train.select_dtypes(include=['number']).columns\ntrain[numerical_cols] = train[numerical_cols].fillna(train[numerical_cols].mean())\n\n\n# Impute missing values using the mode for categorical features\ncategorical_cols = train.select_dtypes(include=['object']).columns\nfor col in categorical_cols:\n    train[col] = train[col].fillna(train[col].mode()[0])","metadata":{"id":"kPGf6NO7T82D","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:43.078382Z","iopub.execute_input":"2024-12-11T05:40:43.078698Z","iopub.status.idle":"2024-12-11T05:40:43.143367Z","shell.execute_reply.started":"2024-12-11T05:40:43.078663Z","shell.execute_reply":"2024-12-11T05:40:43.142468Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Impute missing values using the mean for numerical features\nnumerical_colsT = test.select_dtypes(include=['number']).columns\ntest[numerical_colsT] = test[numerical_colsT].fillna(test[numerical_colsT].mean())\n\n\n# Impute missing values using the mode for categorical features\ncategorical_colsT = test.select_dtypes(include=['object']).columns\nfor col in categorical_colsT:\n    test[col] = test[col].fillna(test[col].mode()[0])","metadata":{"id":"O1bS1T9FWedT","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:43.144471Z","iopub.execute_input":"2024-12-11T05:40:43.144777Z","iopub.status.idle":"2024-12-11T05:40:43.204685Z","shell.execute_reply.started":"2024-12-11T05:40:43.144748Z","shell.execute_reply":"2024-12-11T05:40:43.203917Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 2. Relationship between age and problematic internet use\nplt.figure(figsize=(8, 6))\nsns.boxplot(x=train['sii'], y='Basic_Demos-Age', data=train)\nplt.title('Age vs. Problematic Internet Use')\nplt.show()","metadata":{"id":"isjkMKodeK3i","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:43.205770Z","iopub.execute_input":"2024-12-11T05:40:43.206079Z","iopub.status.idle":"2024-12-11T05:40:43.458363Z","shell.execute_reply.started":"2024-12-11T05:40:43.206050Z","shell.execute_reply":"2024-12-11T05:40:43.457406Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 3. Relationship between BMI and problematic internet use\nplt.figure(figsize=(8, 6))\nsns.boxplot(x=train['sii'], y='Physical-BMI', data=train)\nplt.title('BMI vs. Problematic Internet Use')\nplt.show()","metadata":{"id":"MtLBaB2HkZQM","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:43.459713Z","iopub.execute_input":"2024-12-11T05:40:43.460123Z","iopub.status.idle":"2024-12-11T05:40:43.731613Z","shell.execute_reply.started":"2024-12-11T05:40:43.460079Z","shell.execute_reply":"2024-12-11T05:40:43.730869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 4. Relationship between internet usage hours and problematic internet use\nplt.figure(figsize=(8, 6))\nsns.boxplot(x=train['sii'], y='PreInt_EduHx-computerinternet_hoursday', data=train)\nplt.title('Internet Usage Hours vs. Problematic Internet Use')\nplt.show()","metadata":{"id":"OK72ABqBkZNM","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:43.732759Z","iopub.execute_input":"2024-12-11T05:40:43.733108Z","iopub.status.idle":"2024-12-11T05:40:44.006747Z","shell.execute_reply.started":"2024-12-11T05:40:43.733079Z","shell.execute_reply":"2024-12-11T05:40:44.005792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 6.  Relationship between newly created features and the target variable\n# Example: BMI_Age\nplt.figure(figsize=(8, 6))\nsns.boxplot(x=train['sii'], y='BMI_Age', data=train)\nplt.title('BMI_Age vs. Problematic Internet Use')\nplt.show()","metadata":{"id":"Olf9E2pXlAR0","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:44.008050Z","iopub.execute_input":"2024-12-11T05:40:44.008463Z","iopub.status.idle":"2024-12-11T05:40:44.196733Z","shell.execute_reply.started":"2024-12-11T05:40:44.008412Z","shell.execute_reply":"2024-12-11T05:40:44.195845Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Example: Internet_Hours_Age\nplt.figure(figsize=(8, 6))\nsns.boxplot(x=train['sii'], y='Internet_Hours_Age', data=train)\nplt.title('Internet_Hours_Age vs. Problematic Internet Use')\nplt.show()","metadata":{"id":"mS8aRdLvlCnE","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:44.197841Z","iopub.execute_input":"2024-12-11T05:40:44.198241Z","iopub.status.idle":"2024-12-11T05:40:44.400188Z","shell.execute_reply.started":"2024-12-11T05:40:44.198198Z","shell.execute_reply":"2024-12-11T05:40:44.399223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Distribution of the target variable\nplt.figure(figsize=(8, 6))\nsns.countplot(x=train['sii'], data=train)\nplt.title('Distribution of Target Variable --> sii stands for the severity of problematic internet use')\nplt.show()","metadata":{"id":"lZ-K-2AyXimJ","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:44.401695Z","iopub.execute_input":"2024-12-11T05:40:44.402118Z","iopub.status.idle":"2024-12-11T05:40:44.585154Z","shell.execute_reply.started":"2024-12-11T05:40:44.402069Z","shell.execute_reply":"2024-12-11T05:40:44.584196Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\n# Redefine numeric_cols based on the current columns in the train DataFrame\nnumeric_cols = train.select_dtypes(include=['int32', 'int64', 'float64', 'int64']).columns\ntrain[numeric_cols] = train[numeric_cols].replace([np.inf, -np.inf], np.nan)\ntrain[numeric_cols] = train[numeric_cols].clip(lower=-1e10, upper=1e10)\n\n# Exclude 'sii' from numeric_cols if it's present\nnumeric_cols = numeric_cols.drop('sii', errors='ignore')\n\nscaler = StandardScaler()\ntrain[numeric_cols] = scaler.fit_transform(train[numeric_cols])\n\n# Similarly, redefine numeric_cols for the test DataFrame based on its current columns\nnumeric_cols_test = test.select_dtypes(include=['int32', 'int64', 'float64', 'int64']).columns\n\n# Ensure numeric_cols_test contains the same features as numeric_cols (excluding 'sii')\nnumeric_cols_test = numeric_cols_test.intersection(numeric_cols)\ntest[numeric_cols_test] = test[numeric_cols_test].replace([np.inf, -np.inf], np.nan)\ntest[numeric_cols_test] = test[numeric_cols_test].clip(lower=-1e10, upper=1e10)\n\ntest[numeric_cols_test] = scaler.transform(test[numeric_cols_test])","metadata":{"id":"BsmzGPWzn1d2","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:44.586255Z","iopub.execute_input":"2024-12-11T05:40:44.586514Z","iopub.status.idle":"2024-12-11T05:40:44.724053Z","shell.execute_reply.started":"2024-12-11T05:40:44.586490Z","shell.execute_reply":"2024-12-11T05:40:44.723216Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"id":"hjP8yz8MpkhH","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:44.725232Z","iopub.execute_input":"2024-12-11T05:40:44.725644Z","iopub.status.idle":"2024-12-11T05:40:44.752473Z","shell.execute_reply.started":"2024-12-11T05:40:44.725602Z","shell.execute_reply":"2024-12-11T05:40:44.751654Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head()","metadata":{"id":"etNSZUkLp4sd","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:44.753604Z","iopub.execute_input":"2024-12-11T05:40:44.753911Z","iopub.status.idle":"2024-12-11T05:40:44.780488Z","shell.execute_reply.started":"2024-12-11T05:40:44.753882Z","shell.execute_reply":"2024-12-11T05:40:44.779299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Shape of Train data: \", train.shape)\nprint(\"Shape of Test data: \", test.shape)","metadata":{"id":"mwp_SK0qqzfO","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:44.781780Z","iopub.execute_input":"2024-12-11T05:40:44.782220Z","iopub.status.idle":"2024-12-11T05:40:44.789053Z","shell.execute_reply.started":"2024-12-11T05:40:44.782175Z","shell.execute_reply":"2024-12-11T05:40:44.788006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define constants\nSEED = 42\nn_splits = 5","metadata":{"id":"-zBn0kC7vyk_","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:44.790174Z","iopub.execute_input":"2024-12-11T05:40:44.790485Z","iopub.status.idle":"2024-12-11T05:40:44.801212Z","shell.execute_reply.started":"2024-12-11T05:40:44.790458Z","shell.execute_reply":"2024-12-11T05:40:44.800420Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Quadratic Weighted Kappa metric\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n# Threshold rounding\ndef threshold_rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\n# Evaluate predictions using QWK\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_preds = threshold_rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_preds)","metadata":{"id":"dViDiL4xK5bE","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:44.802257Z","iopub.execute_input":"2024-12-11T05:40:44.802619Z","iopub.status.idle":"2024-12-11T05:40:44.814400Z","shell.execute_reply.started":"2024-12-11T05:40:44.802579Z","shell.execute_reply":"2024-12-11T05:40:44.813413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train and evaluate model using cross-validation\ndef train_and_evaluate(model_class, test_data):\n    # Drop 'id' column from train and test data\n    X = train.drop(['sii', 'id'], axis=1)  # Drop 'id' here\n    y = train['sii']\n    test_data = test_data.drop(['id'], axis=1)  # Drop 'id' here\n\n    skf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    oof_non_rounded = np.zeros(len(y), dtype=float)\n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(skf.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_val_pred = model.predict(X_val)\n        oof_non_rounded[test_idx] = y_val_pred\n\n        test_preds[:, fold] = model.predict(test_data)\n\n    # Optimize thresholds\n    optimal_thresholds = minimize(\n        evaluate_predictions, x0=[0.5, 1.5, 2.5],\n        args=(y, oof_non_rounded), method='Nelder-Mead'\n    )\n    assert optimal_thresholds.success, \"Threshold optimization failed.\"\n\n    # Apply optimized thresholds\n    oof_tuned = threshold_rounder(oof_non_rounded, optimal_thresholds.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"Optimized QWK Score: {tKappa:.4f}\")\n\n    test_preds_mean = test_preds.mean(axis=1)\n    final_test_preds = threshold_rounder(test_preds_mean, optimal_thresholds.x)\n\n    submission = pd.DataFrame({\n        'id': test['id'],  # Use original test data with 'id' for submission\n        'sii': final_test_preds\n    })\n\n    return submission","metadata":{"id":"0Te2IhZDK5O1","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:44.815758Z","iopub.execute_input":"2024-12-11T05:40:44.816056Z","iopub.status.idle":"2024-12-11T05:40:44.828767Z","shell.execute_reply.started":"2024-12-11T05:40:44.816027Z","shell.execute_reply":"2024-12-11T05:40:44.827932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Model parameters\nLightGBM_Params = {\n    'learning_rate': 0.046, 'max_depth': 12, 'num_leaves': 478,\n    'min_data_in_leaf': 13, 'feature_fraction': 0.893, 'bagging_fraction': 0.784,\n    'bagging_freq': 4, 'lambda_l1': 10, 'lambda_l2': 0.01, 'n_estimators': 300\n}\nXGB_Params = {\n    'learning_rate': 0.05, 'max_depth': 6, 'n_estimators': 200, 'subsample': 0.8,\n    'colsample_bytree': 0.8, 'reg_alpha': 1, 'reg_lambda': 5, 'random_state': SEED,\n    'tree_method': 'hist'\n}\nCatBoost_Params = {\n    'learning_rate': 0.05, 'depth': 6, 'iterations': 200, 'random_seed': SEED,\n    'verbose': 0, 'l2_leaf_reg': 10, 'task_type': 'CPU'\n}","metadata":{"id":"5GY4x_j3K5LV","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:44.829829Z","iopub.execute_input":"2024-12-11T05:40:44.830078Z","iopub.status.idle":"2024-12-11T05:40:44.843632Z","shell.execute_reply.started":"2024-12-11T05:40:44.830053Z","shell.execute_reply":"2024-12-11T05:40:44.842791Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize models\nLightGBM_Model = LGBMRegressor(**LightGBM_Params, random_state=SEED, verbose=-1)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)","metadata":{"id":"DUH8otXRLz7s","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:44.844727Z","iopub.execute_input":"2024-12-11T05:40:44.845072Z","iopub.status.idle":"2024-12-11T05:40:44.860936Z","shell.execute_reply.started":"2024-12-11T05:40:44.845020Z","shell.execute_reply":"2024-12-11T05:40:44.860093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Combine models using voting regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', LightGBM_Model),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model)\n], weights=[4.0, 4.0, 5.0])","metadata":{"id":"Up2-sYN9LzuU","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:44.861973Z","iopub.execute_input":"2024-12-11T05:40:44.862309Z","iopub.status.idle":"2024-12-11T05:40:44.871326Z","shell.execute_reply.started":"2024-12-11T05:40:44.862281Z","shell.execute_reply":"2024-12-11T05:40:44.870538Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train and save submission\nsubmission = train_and_evaluate(voting_model, test)\nsubmission.to_csv('submission.csv', index=False)","metadata":{"id":"VM9rs_ZCLzq9","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:44.872378Z","iopub.execute_input":"2024-12-11T05:40:44.872652Z","iopub.status.idle":"2024-12-11T05:41:15.371899Z","shell.execute_reply.started":"2024-12-11T05:40:44.872626Z","shell.execute_reply":"2024-12-11T05:41:15.370960Z"}},"outputs":[],"execution_count":null}]}