{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":7453542,"sourceType":"datasetVersion","datasetId":921302}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-20T13:06:25.155333Z","iopub.execute_input":"2024-12-20T13:06:25.155671Z","iopub.status.idle":"2024-12-20T13:06:31.981869Z","shell.execute_reply.started":"2024-12-20T13:06:25.155636Z","shell.execute_reply":"2024-12-20T13:06:31.980620Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip -q install /kaggle/input/pytorchtabnet/pytorch_tabnet-4.1.0-py3-none-any.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T13:06:31.984043Z","iopub.execute_input":"2024-12-20T13:06:31.984511Z","iopub.status.idle":"2024-12-20T13:07:16.802416Z","shell.execute_reply.started":"2024-12-20T13:06:31.984476Z","shell.execute_reply":"2024-12-20T13:07:16.800714Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Import Library","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\n\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport polars as pl\nimport polars.selectors as cs\n\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom pytorch_tabnet.tab_model import TabNetRegressor\nimport torch\n\npd.options.display.max_columns = None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T13:07:16.804492Z","iopub.execute_input":"2024-12-20T13:07:16.804949Z","iopub.status.idle":"2024-12-20T13:07:36.306633Z","shell.execute_reply.started":"2024-12-20T13:07:16.804909Z","shell.execute_reply":"2024-12-20T13:07:36.305326Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport polars as pl\nimport polars.selectors as cs\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T13:07:36.308011Z","iopub.execute_input":"2024-12-20T13:07:36.308799Z","iopub.status.idle":"2024-12-20T13:07:38.478668Z","shell.execute_reply.started":"2024-12-20T13:07:36.308716Z","shell.execute_reply":"2024-12-20T13:07:38.477399Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score, StratifiedKFold\nimport xgboost as xgb\nimport plotly.express as px","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T13:07:38.480081Z","iopub.execute_input":"2024-12-20T13:07:38.480770Z","iopub.status.idle":"2024-12-20T13:07:39.110284Z","shell.execute_reply.started":"2024-12-20T13:07:38.480731Z","shell.execute_reply":"2024-12-20T13:07:39.109075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from lightgbm import LGBMRegressor\n\nfrom catboost import CatBoostRegressor  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T13:07:39.111878Z","iopub.execute_input":"2024-12-20T13:07:39.112417Z","iopub.status.idle":"2024-12-20T13:07:39.119409Z","shell.execute_reply.started":"2024-12-20T13:07:39.112362Z","shell.execute_reply":"2024-12-20T13:07:39.117894Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Processing Data","metadata":{}},{"cell_type":"code","source":"path = '../input/child-mind-institute-problematic-internet-use/'\ntrain = pd.read_csv(path + 'train.csv', index_col = 'id')\nprint(\"The train data has the shape: \",train.shape)\ntest = pd.read_csv(path + 'test.csv', index_col = 'id')\nprint(\"The test data has the shape: \",test.shape)\nprint(\"\")\nprint(\"Total number of missing training values: \", train.isna().sum().sum())\ntrain\ntest","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T13:07:39.123110Z","iopub.execute_input":"2024-12-20T13:07:39.123519Z","iopub.status.idle":"2024-12-20T13:07:39.323147Z","shell.execute_reply.started":"2024-12-20T13:07:39.123479Z","shell.execute_reply":"2024-12-20T13:07:39.322013Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Features","metadata":{}},{"cell_type":"code","source":"# Preprocessing các column \"season\" có trong train, fill nan với giá trị 0\ntrain_cat_columns = train.select_dtypes(exclude='number').columns\nprint(train_cat_columns)\n\nfor season in train_cat_columns:\n    train[season] = train[season].fillna(0)\n    train[season] = train[season].replace({'Spring': 1, 'Summer': 2, 'Fall': 3, 'Winter': 4}).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T13:07:39.324715Z","iopub.execute_input":"2024-12-20T13:07:39.325235Z","iopub.status.idle":"2024-12-20T13:07:39.376477Z","shell.execute_reply.started":"2024-12-20T13:07:39.325171Z","shell.execute_reply":"2024-12-20T13:07:39.375238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Preprocessing các column \"season\" có trong train, fill nan với giá trị 0\ntest_cat_columns = test.select_dtypes(exclude='number').columns\nprint(test_cat_columns)\n\nfor season in test_cat_columns:\n    test[season] = test[season].fillna(0)\n    test[season] = test[season].replace({'Spring': 1, 'Summer': 2, 'Fall': 3, 'Winter': 4}).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T13:07:39.378338Z","iopub.execute_input":"2024-12-20T13:07:39.378673Z","iopub.status.idle":"2024-12-20T13:07:39.400767Z","shell.execute_reply.started":"2024-12-20T13:07:39.378642Z","shell.execute_reply":"2024-12-20T13:07:39.399189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PCIAT_cols = [val for val in train.columns[train.columns.str.contains('PCIAT')]]\nprint('Number of PCIAT features = ' , len(PCIAT_cols))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T13:07:39.402467Z","iopub.execute_input":"2024-12-20T13:07:39.403043Z","iopub.status.idle":"2024-12-20T13:07:39.414699Z","shell.execute_reply.started":"2024-12-20T13:07:39.402988Z","shell.execute_reply":"2024-12-20T13:07:39.413221Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig = px.scatter(train, x = 'PCIAT-PCIAT_Total', color = 'sii', marginal_x=\"box\", title = 'PCIAT Total')\nfig = fig.update_layout(yaxis_title=\"\")\nfig.update_yaxes(showticklabels=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T13:07:39.416291Z","iopub.execute_input":"2024-12-20T13:07:39.416742Z","iopub.status.idle":"2024-12-20T13:07:41.765165Z","shell.execute_reply.started":"2024-12-20T13:07:39.416688Z","shell.execute_reply":"2024-12-20T13:07:41.763959Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.sii.value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T13:07:41.766899Z","iopub.execute_input":"2024-12-20T13:07:41.767355Z","iopub.status.idle":"2024-12-20T13:07:41.783541Z","shell.execute_reply.started":"2024-12-20T13:07:41.767303Z","shell.execute_reply":"2024-12-20T13:07:41.781962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PCIAT_cols.remove('PCIAT-PCIAT_Total') #Column này sử dụng để làm giá trị Ground truth cho train\ntrain = train.drop(columns = PCIAT_cols) #Test không có các column PCIAT nên drop để về đúng format.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T13:07:41.785631Z","iopub.execute_input":"2024-12-20T13:07:41.786297Z","iopub.status.idle":"2024-12-20T13:07:41.807733Z","shell.execute_reply.started":"2024-12-20T13:07:41.786233Z","shell.execute_reply":"2024-12-20T13:07:41.806424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train.shape)\ntrain","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T13:07:41.809349Z","iopub.execute_input":"2024-12-20T13:07:41.809709Z","iopub.status.idle":"2024-12-20T13:07:41.888899Z","shell.execute_reply.started":"2024-12-20T13:07:41.809675Z","shell.execute_reply":"2024-12-20T13:07:41.887609Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Plot Data","metadata":{}},{"cell_type":"code","source":"sns.countplot(train, x = 'sii').set_title('Count of sii') ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T13:07:41.890658Z","iopub.execute_input":"2024-12-20T13:07:41.891166Z","iopub.status.idle":"2024-12-20T13:07:42.243191Z","shell.execute_reply.started":"2024-12-20T13:07:41.891114Z","shell.execute_reply":"2024-12-20T13:07:42.241856Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Plot sii theo thời gian sử dụng Internet\nvals = ['PIU = 0', 'PIU = 1','PIU = 2', 'PIU = 3']\n\nfor i in range(4):\n    plt.figure()\n    plot = sns.countplot(x = train[train.sii==i]['PreInt_EduHx-computerinternet_hoursday'])\n    plot.set_title(vals[i])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T13:07:42.244650Z","iopub.execute_input":"2024-12-20T13:07:42.245054Z","iopub.status.idle":"2024-12-20T13:07:43.078805Z","shell.execute_reply.started":"2024-12-20T13:07:42.245017Z","shell.execute_reply":"2024-12-20T13:07:43.077342Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.dropna(subset='sii')\ntrain","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T13:07:43.080679Z","iopub.execute_input":"2024-12-20T13:07:43.081232Z","iopub.status.idle":"2024-12-20T13:07:43.155446Z","shell.execute_reply.started":"2024-12-20T13:07:43.081180Z","shell.execute_reply":"2024-12-20T13:07:43.154084Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Correlation","metadata":{}},{"cell_type":"code","source":"#Đánh gái mức độ tương quan giữa các column với ground truth\ncorr = pd.DataFrame(train.corr()['PCIAT-PCIAT_Total'].sort_values(ascending = False))\ncorr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T13:07:43.157309Z","iopub.execute_input":"2024-12-20T13:07:43.157689Z","iopub.status.idle":"2024-12-20T13:07:43.200603Z","shell.execute_reply.started":"2024-12-20T13:07:43.157653Z","shell.execute_reply":"2024-12-20T13:07:43.199465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Chọn ra các features có corr cao với PCIAT_Total và loại bớt các feat có độ tương đồng thấp hoặc tương đương với các feat đã chọn\nselection = corr[(corr['PCIAT-PCIAT_Total']>0.1) | (corr['PCIAT-PCIAT_Total']<-0.1)]\nselection = [val for val in selection.index]\nselection.remove('PCIAT-PCIAT_Total')\nselection.remove('sii')\nselection.remove('Physical-BMI')\nselection.remove('SDS-SDS_Total_Raw')\nselection.remove('FGC-FGC_SRR_Zone')\nselection.remove('FGC-FGC_SRL_Zone')\nselection","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T13:07:43.202188Z","iopub.execute_input":"2024-12-20T13:07:43.202538Z","iopub.status.idle":"2024-12-20T13:07:43.213640Z","shell.execute_reply.started":"2024-12-20T13:07:43.202505Z","shell.execute_reply":"2024-12-20T13:07:43.212252Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Actigraphy (time series data)","metadata":{}},{"cell_type":"code","source":"actigraphy = pl.read_parquet('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id=0417c91e/part-0.parquet')\nactigraphy","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T13:07:43.215569Z","iopub.execute_input":"2024-12-20T13:07:43.216104Z","iopub.status.idle":"2024-12-20T13:07:43.455692Z","shell.execute_reply.started":"2024-12-20T13:07:43.216056Z","shell.execute_reply":"2024-12-20T13:07:43.454531Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom scipy.stats import linregress\n\ndef transform_time_of_day(df):\n    # TimeofDay đang ở ms nên ta sẽ đổi sang giờ cho dễ tính và nhỏ hơn\n    df['hour'] = (df['time_of_day'] % (24 * 60 * 60 * 1000)) / (60 * 60 * 1000)\n    return df\n\ndef impute_missing_values(features):\n    # Điền các giá trị còn thiếu cho từng cột\n    for col in features:\n        # Cột nào có tên mean thì điền theo mean\n        if \"mean\" in col or \"avg\" in col:\n            features[col].fillna(features[col].mean(), inplace=True)\n        elif \"std\" in col or \"variation\" in col:\n            # Khi độ lệch chuẩn trong cột hay 'std' quá nhỏ thì ta điền giá trị còn thiếu = 0\n            features[col].fillna(0, inplace=True)\n        elif \"spike_count\" in col or \"count\" in col:\n            # Điền giá trị spike count bằng 0, giả định không có spike nếu thiếu\n            features[col].fillna(0, inplace=True)\n        else:\n            # Không trong trường hợp nào thì điền = med\n            features[col].fillna(features[col].median(), inplace=True)\n    return features\n\ndef extract_time_series_features(actigraphy_df):\n    # Biến đổi thành hour\n    actigraphy_df = transform_time_of_day(actigraphy_df)\n    \n    night_hours = (0, 6)\n\n    features = {}\n\n    # ENMO Features\n    features['weekly_enmo_mean'] = actigraphy_df['enmo'].mean()\n    features['weekly_enmo_std'] = actigraphy_df['enmo'].std()\n    features['night_enmo_mean'] = actigraphy_df[(actigraphy_df['hour'] >= night_hours[0]) & (actigraphy_df['hour'] < night_hours[1])]['enmo'].mean()\n\n    # Tạo ra các features ENMO mới dựa theo time series data\n    actigraphy_df['enmo_ma_2h'] = actigraphy_df['enmo'].rolling(window=120, min_periods=1).mean()\n    actigraphy_df['enmo_ma_6h'] = actigraphy_df['enmo'].rolling(window=360, min_periods=1).mean()\n    features['enmo_2h_ma_avg'] = actigraphy_df['enmo_ma_2h'].mean()\n    features['enmo_6h_ma_avg'] = actigraphy_df['enmo_ma_6h'].mean()\n\n    features['enmo_gradient_mean'] = actigraphy_df['enmo'].diff().mean()\n    features['enmo_spike_count'] = (actigraphy_df['enmo'].diff() > 0.1).sum()  \n\n    actigraphy_df['quarter'] = (actigraphy_df['time_of_day'] // (3 * 30 * 24 * 60 * 60 * 1000)) % 4 + 1\n    quarterly_enmo = actigraphy_df.groupby('quarter')['enmo'].mean()\n    features['enmo_q1_avg'] = quarterly_enmo.get(1, np.nan)\n    features['enmo_q2_avg'] = quarterly_enmo.get(2, np.nan)\n    features['enmo_q3_avg'] = quarterly_enmo.get(3, np.nan)\n    features['enmo_q4_avg'] = quarterly_enmo.get(4, np.nan)\n    features['enmo_seasonal_variation'] = quarterly_enmo.std()\n\n    # Features Light based\n    night_light = actigraphy_df[(actigraphy_df['hour'] >= night_hours[0]) & (actigraphy_df['hour'] < night_hours[1])]\n    features['avg_night_light'] = night_light['light'].mean()\n    features['night_light_spike_count'] = (night_light['light'].diff() > 0.2).sum()  \n    \n    # Features Battery based \n    battery_drop_count = (actigraphy_df['battery_voltage'].diff() < 0).sum()\n    features['battery_drop_count'] = battery_drop_count\n\n    # Percentage Non-Wear\n    features['non_wear_percentage'] = actigraphy_df['non-wear_flag'].mean()\n    \n    # AngleZ\n    actigraphy_df['anglez_grad'] = actigraphy_df['anglez'].diff()\n    features['avg_anglez'] = actigraphy_df['anglez'].mean()\n    features['anglez_std'] = actigraphy_df['anglez'].std()\n    features['posture_deterioration_spikes'] = (actigraphy_df['anglez_grad'].abs() > 5).sum()  # Đặt ngưỡng của spike là 5 \n    \n    # Phân tích xu hướng\n    trend_enmo = linregress(range(len(actigraphy_df)), actigraphy_df['enmo'])[0]  \n    trend_light = linregress(range(len(actigraphy_df)), actigraphy_df['light'])[0]  \n    features['enmo_trend_slope'] = trend_enmo\n    features['light_trend_slope'] = trend_light\n\n    # Features tạm thời cho hoạt động\n    features['nighttime_activity_proportion'] = night_light['enmo'].sum() / actigraphy_df['enmo'].sum()\n    \n    # Hoạt động trong ngày thường so với cuối tuần\n    weekday_enmo = actigraphy_df[actigraphy_df['weekday'] < 5]['enmo'].mean()\n    weekend_enmo = actigraphy_df[actigraphy_df['weekday'] >= 5]['enmo'].mean()\n    features['weekday_enmo_avg'] = weekday_enmo\n    features['weekend_enmo_avg'] = weekend_enmo\n\n    \n    features_df = pd.DataFrame(features, index=[0])\n    features_df = impute_missing_values(features_df)\n    return features_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T13:07:43.457157Z","iopub.execute_input":"2024-12-20T13:07:43.457519Z","iopub.status.idle":"2024-12-20T13:07:43.478545Z","shell.execute_reply.started":"2024-12-20T13:07:43.457485Z","shell.execute_reply":"2024-12-20T13:07:43.477335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\nfrom tqdm import tqdm\nfrom concurrent.futures import ThreadPoolExecutor\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    \n    features_df = extract_time_series_features(df)\n    \n    # Đổi thành vector 1 chiều\n    describe_df = df.describe().reset_index(drop=True)\n    describe_flattened = describe_df.values.flatten()\n    \n    describe_features = pd.DataFrame([describe_flattened], columns=[f\"{col}_{stat}\" for col in df.columns for stat in describe_df.index])\n    combined_features_df = pd.concat([features_df, describe_features], axis=1)\n    \n    # Đổi combined_features_df thành vector 1 chiều\n    features_flat = combined_features_df.values.flatten()\n    \n    return features_flat, filename.split('=')[1] # Trả về cả ID\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])  # Each row corresponds to a file's features\n    df['id'] = indexes \n    \n    return df\n\n\nclass AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim*3),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*3, encoding_dim*2),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*2, encoding_dim),\n            nn.ReLU()\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim*2),\n            nn.ReLU(),\n            nn.Linear(input_dim*2, input_dim*3),\n            nn.ReLU(),\n            nn.Linear(input_dim*3, input_dim),\n            nn.Sigmoid()\n        )\n        \n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded\n\n\ndef perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    \n    data_tensor = torch.FloatTensor(df_scaled)\n    \n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n    \n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(autoencoder.parameters())\n    \n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n            \n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n                 \n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n        \n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n    \n    return df_encoded\n\ndef feature_engineering(df):\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1) \n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T13:07:43.484509Z","iopub.execute_input":"2024-12-20T13:07:43.484961Z","iopub.status.idle":"2024-12-20T13:07:43.872394Z","shell.execute_reply.started":"2024-12-20T13:07:43.484924Z","shell.execute_reply":"2024-12-20T13:07:43.871115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\n# Hàm để chuẩn hóa dữ liệu\ndef scale_data(df):\n    \"\"\"Chuẩn hóa dữ liệu bằng StandardScaler và trả về dữ liệu đã chuẩn hóa cùng với scaler.\"\"\"\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    return df_scaled, scaler\n\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\n# Chuẩn bị dữ liệu chuỗi thời gian và loại bỏ cột 'id' để chuẩn hóa\ndf_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\n# Xác định các cột không thay đổi\nconstant_columns = [col for col in df_train.columns if df_train[col].nunique() == 1]\n\n# Loại bỏ các cột không thay đổi từ tập train và test\ndf_train = df_train.drop(columns=constant_columns)\ndf_test = df_test.drop(columns=constant_columns)\n\nsmall_variance_columns = df_train.columns[df_train.std() < 1e-5]  # Điều chỉnh ngưỡng nếu cần\nfor col in small_variance_columns:\n    df_train[col] = df_train[col].fillna(df_train[col].mean())\n    df_test[col] = df_test[col].fillna(df_test[col].mean())\n\n# Chuẩn hóa dữ liệu huấn luyện và kiểm tra riêng biệt\ndf_train_scaled, train_scaler = scale_data(df_train)\ndf_test_scaled = train_scaler.transform(df_test)\n\n# Chuyển đổi dữ liệu đã chuẩn hóa thành DataFrames\ndf_train_scaled = pd.DataFrame(df_train_scaled, columns=df_train.columns)\ndf_test_scaled = pd.DataFrame(df_test_scaled, columns=df_test.columns)\ndf_train_scaled.fillna(0, inplace=True)\ndf_test_scaled.fillna(0, inplace=True)\n\n# Check lại xem còn NaNs không\nprint(\"NaNs trong dữ liệu train đã chuẩn hóa:\", df_train_scaled.isna().sum().sum())\nprint(\"NaNs trong dữ liệu test đã chuẩn hóa:\", df_test_scaled.isna().sum().sum())\n\n# Thực hiện mã hóa autoencoder trên dữ liệu đã chuẩn hóa\ntrain_ts_encoded = perform_autoencoder(df_train_scaled, encoding_dim=45, epochs=100, batch_size=32)\ntest_ts_encoded = perform_autoencoder(df_test_scaled, encoding_dim=45, epochs=100, batch_size=32)\n\ntime_series_cols = train_ts_encoded.columns.tolist()\n\n# Thêm cột 'id' trở lại dữ liệu đã mã hóa để gộp\ntrain_ts_encoded[\"id\"] = train_ts[\"id\"]\ntest_ts_encoded[\"id\"] = test_ts[\"id\"]\n\n# Gộp các đặc trưng chuỗi thời gian đã mã hóa trở lại với dữ liệu chính của train và test\ntrain = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')\n\n# Điền giá trị còn thiếu\nimputer = KNNImputer(n_neighbors=5)\nnumeric_cols = train.select_dtypes(include=['float64', 'int64']).columns\nimputed_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\n        \ntrain = train_imputed\n\ntrain = feature_engineering(train)\ntrain = train.dropna(thresh=10, axis=0)\ntest = feature_engineering(test)\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\n# Định nghĩa và lọc các cột đặc trưng đã chọn\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD',\n                'FGC-FGC_GSD_Zone', 'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', \n                'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-BIA_Activity_Level_num', \n                'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', \n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num', 'BIA-BIA_ICW', 'BIA-BIA_LDM', \n                'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total', \n                'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', 'PreInt_EduHx-computerinternet_hoursday', 'sii', \n                'BMI_Age', 'Internet_Hours_Age', 'BMI_Internet_Hours', 'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', \n                'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight', 'SMM_Height', 'Muscle_to_Fat', \n                'Hydration_Status', 'ICW_TBW'] + time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ntest = test[[c for c in featuresCols if c !=\"sii\"]]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T13:07:43.873985Z","iopub.execute_input":"2024-12-20T13:07:43.874387Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if np.any(np.isinf(train)):\n    train = train.replace([np.inf, -np.inf], np.nan)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train Model\n","metadata":{}},{"cell_type":"code","source":"def TrainModel(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=5, shuffle=True, random_state=2024)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), 5))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=5)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Hyperparameter Tuning\n","metadata":{}},{"cell_type":"code","source":"# Tinh chỉnh siêu tham số bằng gridSearch\n\"\"\"\nfrom scipy.stats import uniform, randint\n\nfrom sklearn.model_selection import GridSearchCV\n\ndef grid_search_hyperparameter_tuning(X, y):\n    # Định nghĩa grid các siêu tham số cho từng mô hình\n    param_grids = {\n        'xgb': {\n            'max_depth': randint(4, 8),              # Bao gồm 6\n            'n_estimators': randint(100, 200),       \n            'learning_rate': uniform(0.03, 0.04),    \n            'subsample': uniform(0.5, 0.2),          \n            'colsample_bytree': uniform(0.7, 0.2),   \n            'reg_alpha': uniform(0.5, 1.5),          \n            'reg_lambda': uniform(3, 4), \n        },\n        'cat': {\n            'iterations': randint(350, 451),\n            'learning_rate': uniform(0.02, 0.02),    \n            'depth': randint(5, 8),                  \n            'l2_leaf_reg': uniform(15, 10),          \n            'random_seed': [2024]\n        },\n        'lgbm': {\n            'learning_rate': uniform(0.04, 0.01),     \n            'max_depth': randint(8, 11),              \n            'num_leaves': randint(450, 501),          \n            'min_data_in_leaf': randint(10, 16),      \n            'feature_fraction': uniform(0.8, 0.2),     \n            'bagging_fraction': uniform(0.7, 0.2),     \n            'bagging_freq': randint(3, 6),            \n            'lambda_l1': uniform(8, 4),               \n            'lambda_l2': uniform(0.005, 0.01),\n            'verbose': -1\n        }\n    }\n    \n    # Thiết lập cross-validation\n    cv = StratifiedKFold(n_splits=5, shuffle=True, random_state=2024)\n    \n    # Khởi tạo các mô hình\n    models = {\n        'xgb': xgb.XGBRegressor(random_state=2024),\n        'cat': CatBoostRegressor(silent=True, random_seed=2024),\n        'lgbm': LGBMRegressor(random_state=2024, force_col_wise=True, verbosity=-1)\n    }\n    \n    # Lưu trữ các mô hình tốt nhất\n    best_models = {}\n    \n    # Thực hiện Grid Search cho từng mô hình\n    for name, model in models.items():\n        grid_search = GridSearchCV(\n            estimator=model,\n            param_grid=param_grids[name],\n            scoring=kappa_scorer,  # Sử dụng custom scoring\n            cv=cv,\n            n_jobs=-1,  # Sử dụng tất cả các nhân CPU\n            verbose=2\n        )\n        \n        # Fit grid search\n        grid_search.fit(X, y)\n        \n        # Lưu mô hình tốt nhất\n        best_models[name] = grid_search.best_estimator_\n        \n        # In ra thông tin về grid search\n        print(f\"\\n{name.upper()} Best Parameters:\")\n        print(grid_search.best_params_)\n        print(f\"{name.upper()} Best Score: {grid_search.best_score_}\")\n    \n    return best_models\n\"\"\"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"XGB_Params ={\n    'max_depth': 6,\n    'n_estimators': 150,\n    'learning_rate': 0.05,\n    'subsample': 0.6,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,\n    'reg_lambda': 5,\n}\n\nCAT_Params = {\n    'iterations': 380,\n    'learning_rate': 0.03,\n    'depth': 6,\n    'l2_leaf_reg': 20,\n    'random_seed': 2024,\n}\n\nlgb_Params = {\n    'learning_rate': 0.046,\n    'max_depth': 9,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.9,\n    'bagging_fraction': 0.8,\n    'bagging_freq': 4,\n    'verbose': -1,\n    'lambda_l1': 10,\n    'lambda_l2': 0.01,\n}","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Model Tabnet\n\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import train_test_split\nfrom pytorch_tabnet.callbacks import Callback\nimport os\nimport torch\nfrom pytorch_tabnet.callbacks import Callback\n\nclass TabNetWrapper(BaseEstimator, RegressorMixin):\n    def __init__(self, **kwargs):\n        self.model = TabNetRegressor(**kwargs)\n        self.kwargs = kwargs\n        self.imputer = SimpleImputer(strategy='median')\n        self.best_model_path = 'best_tabnet_model.pt'\n        \n    def fit(self, X, y):\n        # Handle missing values\n        X_imputed = self.imputer.fit_transform(X)\n        \n        if hasattr(y, 'values'):\n            y = y.values\n            \n        # Create internal validation set\n        X_train, X_valid, y_train, y_valid = train_test_split(\n            X_imputed, \n            y, \n            test_size=0.2,\n            random_state=42\n        )\n        \n        # Train TabNet model\n        history = self.model.fit(\n            X_train=X_train,\n            y_train=y_train.reshape(-1, 1),\n            eval_set=[(X_valid, y_valid.reshape(-1, 1))],\n            eval_name=['valid'],\n            eval_metric=['mse'],\n            max_epochs=200,\n            patience=20,\n            batch_size=1024,\n            virtual_batch_size=128,\n            num_workers=0,\n            drop_last=False,\n            callbacks=[\n                TabNetPretrainedModelCheckpoint(\n                    filepath=self.best_model_path,\n                    monitor='valid_mse',\n                    mode='min',\n                    save_best_only=True,\n                    verbose=True\n                )\n            ]\n        )\n        \n        # Load the best model\n        if os.path.exists(self.best_model_path):\n            self.model.load_model(self.best_model_path)\n            os.remove(self.best_model_path)  \n        \n        return self\n    \n    def predict(self, X):\n        X_imputed = self.imputer.transform(X)\n        return self.model.predict(X_imputed).flatten()\n    \n    def __deepcopy__(self, memo):\n        cls = self.__class__\n        result = cls.__new__(cls)\n        memo[id(self)] = result\n        for k, v in self.__dict__.items():\n            setattr(result, k, deepcopy(v, memo))\n        return result\n\n# TabNet siêu tham số\nTabNet_Params = {\n    'n_d': 64,              \n    'n_a': 64,              \n    'n_steps': 5,           \n    'gamma': 1.5,           \n    'n_independent': 2,    \n    'n_shared': 2,          \n    'lambda_sparse': 1e-4,  \n    'optimizer_fn': torch.optim.Adam,\n    'optimizer_params': dict(lr=2e-2, weight_decay=1e-5),\n    'mask_type': 'entmax',\n    'scheduler_params': dict(mode=\"min\", patience=10, min_lr=1e-5, factor=0.5),\n    'scheduler_fn': torch.optim.lr_scheduler.ReduceLROnPlateau,\n    'verbose': 1,\n    'device_name': 'cuda' if torch.cuda.is_available() else 'cpu'\n}\n\nclass TabNetPretrainedModelCheckpoint(Callback):\n    def __init__(self, filepath, monitor='val_loss', mode='min', \n                 save_best_only=True, verbose=1):\n        super().__init__() \n        self.filepath = filepath\n        self.monitor = monitor\n        self.mode = mode\n        self.save_best_only = save_best_only\n        self.verbose = verbose\n        self.best = float('inf') if mode == 'min' else -float('inf')\n        \n    def on_train_begin(self, logs=None):\n        self.model = self.trainer \n        \n    def on_epoch_end(self, epoch, logs=None):\n        logs = logs or {}\n        current = logs.get(self.monitor)\n        if current is None:\n            return\n        \n        # Check if current metric is better than best\n        if (self.mode == 'min' and current < self.best) or \\\n           (self.mode == 'max' and current > self.best):\n            if self.verbose:\n                print(f'\\nEpoch {epoch}: {self.monitor} improved from {self.best:.4f} to {current:.4f}')\n            self.best = current\n            if self.save_best_only:\n                self.model.save_model(self.filepath) ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Model","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgb_model = LGBMRegressor(**lgb_Params, random_state=2024, n_estimators=300)\nxgb_model = xgb.XGBRegressor(**XGB_Params)\ncat_model = CatBoostRegressor(**CAT_Params, silent=True)\n# rf_model = RandomForestRegressor(**RF_Params)\nTabNet_Model = TabNetWrapper(**TabNet_Params)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model 1","metadata":{}},{"cell_type":"code","source":"voting_model = VotingRegressor(estimators=[\n    ('lgb', lgb_model),\n    ('xgb', xgb_model),\n    ('cat', cat_model),\n    ('tabnet', TabNet_Model)\n    # ('rf', rf_model)\n], weights=[4.0,4.0,5.0,4.0])\n\nSub1 = TrainModel(voting_model, test)\n\nSub1","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model 2","metadata":{}},{"cell_type":"code","source":"# Chạy thêm model ensemble khác với LGB, XGB, CB\n# Làm tương tự với Model 1\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n        \ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainModel2(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=5, shuffle=True, random_state=2024)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), 5))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=5)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission\n\n# Siêu tham số cho các mô hình\nXGB_Params ={\n    'max_depth': 6,\n    'n_estimators': 150,\n    'learning_rate': 0.05,\n    'subsample': 0.6,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,\n    'reg_lambda': 5,\n}\n\nCAT_Params = {\n    'iterations': 380,\n    'learning_rate': 0.03,\n    'depth': 6,\n    'l2_leaf_reg': 20,\n    'random_seed': 2024,\n}\n\nlgb_Params = {\n    'learning_rate': 0.046,\n    'max_depth': 9,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.9,\n    'bagging_fraction': 0.8,\n    'bagging_freq': 4,\n    'lambda_l1': 10,\n    'lambda_l2': 0.01,\n}\n\n# Create model instances\nlgb_model = LGBMRegressor(**lgb_Params, random_state=2024, verbose=-1, n_estimators=300)\nxgb_model = XGBRegressor(**XGB_Params)\ncat_model = CatBoostRegressor(**CAT_Params)\n\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', lgb_model),\n    ('xgboost', xgb_model),\n    ('catboost', cat_model)\n])\n\n# Train the ensemble model\nSub2 = TrainModel2(voting_model, test)\n\n# Save submission\n#Submission2.to_csv('submission.csv', index=False)\nSub2","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model 3","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainModel3(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=5, shuffle=True, random_state=2024)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), 5))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=5)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tp_rounded = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n    return tp_rounded\n\nimputer = SimpleImputer(strategy='median')\n\nensemble = VotingRegressor(estimators=[\n    ('lgb', Pipeline(steps=[('imputer', imputer), ('regressor', LGBMRegressor(random_state=2024))])),\n    ('xgb', Pipeline(steps=[('imputer', imputer), ('regressor', XGBRegressor(random_state=2024))])),\n    ('cat', Pipeline(steps=[('imputer', imputer), ('regressor', CatBoostRegressor(random_state=2024, silent=True))])),\n    ('rf', Pipeline(steps=[('imputer', imputer), ('regressor', RandomForestRegressor(random_state=2024))])),\n    ('gb', Pipeline(steps=[('imputer', imputer), ('regressor', GradientBoostingRegressor(random_state=2024))]))\n])\n\nSub3 = TrainModel3(ensemble, test)\nSub3 = pd.DataFrame({\n    'id': sample['id'],\n    'sii': Sub3\n})\n\nSub3","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Submission","metadata":{}},{"cell_type":"code","source":"sub1 = Sub1\nsub2 = Sub2\nsub3 = Sub3\n\nsub1 = sub1.sort_values(by='id').reset_index(drop=True)\nsub2 = sub2.sort_values(by='id').reset_index(drop=True)\nsub3 = sub3.sort_values(by='id').reset_index(drop=True)\n\ncombined = pd.DataFrame({\n    'id': sub1['id'],\n    'sii_1': sub1['sii'],\n    'sii_2': sub2['sii'],\n    'sii_3': sub3['sii']\n})\n\ndef majority_vote(row):\n    return row.mode()[0]\n\ncombined['final_sii'] = combined[['sii_1', 'sii_2', 'sii_3']].apply(majority_vote, axis=1)\n\nfinal_submission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\n\nfinal_submission.to_csv('submission.csv', index=False)\n\nfinal_submission","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}