{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestClassifier, RandomForestRegressor\nfrom sklearn.model_selection import train_test_split,cross_val_score\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nfrom sklearn.impute import SimpleImputer, KNNImputer\nimport polars as pl\nimport lightgbm as lgbm\nfrom lightgbm import LGBMClassifier, LGBMRegressor\nimport catboost as cb\nfrom catboost import CatBoostClassifier, CatBoostRegressor\nimport xgboost as xgb\nfrom xgboost import XGBClassifier, XGBRegressor\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import accuracy_score, classification_report\nfrom sklearn.metrics import make_scorer, cohen_kappa_score\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom keras.optimizers import Adam\nfrom sklearn.preprocessing import StandardScaler\nimport torch.optim as optim\nfrom sklearn.ensemble import VotingRegressor, StackingRegressor\nfrom sklearn.linear_model import Ridge\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor, HistGradientBoostingRegressor\nimport seaborn as sns\n\n%matplotlib inline\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\nSEED = 42\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true,"execution":{"iopub.status.busy":"2024-12-22T11:57:38.912350Z","iopub.execute_input":"2024-12-22T11:57:38.912724Z","iopub.status.idle":"2024-12-22T11:57:59.616229Z","shell.execute_reply.started":"2024-12-22T11:57:38.912682Z","shell.execute_reply":"2024-12-22T11:57:59.615136Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Version 17:\nThêm encoder cho dữ liệu time_series\nImputer các cột season trước bước drop dưới đây\nDrop: các cột có nhiều hơn 50% giá trị thiếu -> đổi thành 30%\nChuyển đổi các giá trị Spring = 1, ...","metadata":{}},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"# Đọc dữ liệu\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsub_sample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T11:57:59.619027Z","iopub.execute_input":"2024-12-22T11:57:59.619954Z","iopub.status.idle":"2024-12-22T11:57:59.703201Z","shell.execute_reply.started":"2024-12-22T11:57:59.619906Z","shell.execute_reply":"2024-12-22T11:57:59.701883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train.sii.value_counts())\nsns.countplot(train, x = 'sii').set_title('Count of sii')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T11:57:59.704955Z","iopub.execute_input":"2024-12-22T11:57:59.705423Z","iopub.status.idle":"2024-12-22T11:58:00.041625Z","shell.execute_reply.started":"2024-12-22T11:57:59.705375Z","shell.execute_reply":"2024-12-22T11:58:00.040408Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"season_cols = [col for col in test if 'Season' in col]\n\nplt.figure(figsize = (10,20))\nfor i, col in enumerate(season_cols, 1):\n    plt.subplot(5, 2, i)  # 5 rows, 2 columns, plot i\n    sns.boxplot(x=col, y='sii', data=train)\n    plt.xticks(rotation = 45)\n    plt.title(f\"'sii' vs {col}\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T11:58:00.043126Z","iopub.execute_input":"2024-12-22T11:58:00.043592Z","iopub.status.idle":"2024-12-22T11:58:02.623161Z","shell.execute_reply.started":"2024-12-22T11:58:00.043542Z","shell.execute_reply":"2024-12-22T11:58:02.621586Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Xác suất các mùa\nseason_probabilities = train['Basic_Demos-Enroll_Season'].value_counts(normalize=True)\nseason_probabilities","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T11:58:02.624650Z","iopub.execute_input":"2024-12-22T11:58:02.625032Z","iopub.status.idle":"2024-12-22T11:58:02.635160Z","shell.execute_reply.started":"2024-12-22T11:58:02.624996Z","shell.execute_reply":"2024-12-22T11:58:02.634095Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nprint('Các cột bị thiếu trong test data:')\nprint([f for f in train.columns if f not in test.columns])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T11:58:02.636536Z","iopub.execute_input":"2024-12-22T11:58:02.636928Z","iopub.status.idle":"2024-12-22T11:58:02.647252Z","shell.execute_reply.started":"2024-12-22T11:58:02.636896Z","shell.execute_reply":"2024-12-22T11:58:02.646256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Tính toán tỷ lệ phần trăm giá trị thiếu cho từng cột\nmissing_percentage = train.isnull().mean() * 100\nmissing_percentage = missing_percentage[missing_percentage > 0].sort_values(ascending=False)  # Lọc các cột có giá trị thiếu\n\n# Vẽ biểu đồ cột\nplt.figure(figsize=(10, 25))\nsns.barplot(x=missing_percentage.values, y=missing_percentage.index, palette=\"viridis\")\nplt.title(\"Tỷ lệ phần trăm giá trị thiếu theo từng cột\", fontsize=16)\nplt.xlabel(\"Tỷ lệ phần trăm (%)\", fontsize=14)\nplt.ylabel(\"Tên cột\", fontsize=14)\nplt.grid(axis='x', linestyle='--', alpha=0.7)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T11:58:02.650828Z","iopub.execute_input":"2024-12-22T11:58:02.651424Z","iopub.status.idle":"2024-12-22T11:58:03.782903Z","shell.execute_reply.started":"2024-12-22T11:58:02.651366Z","shell.execute_reply":"2024-12-22T11:58:03.781650Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"    sns.boxplot(x='sii', y='PCIAT-PCIAT_Total', data=train)\n    plt.title(f'Mối quan hệ giữa PCIAT-PCIAT_Total và sii')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T11:58:03.784282Z","iopub.execute_input":"2024-12-22T11:58:03.784634Z","iopub.status.idle":"2024-12-22T11:58:04.035910Z","shell.execute_reply.started":"2024-12-22T11:58:03.784601Z","shell.execute_reply":"2024-12-22T11:58:04.034830Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_numeric = train.select_dtypes(include=['number'])\n\n# Tính toán ma trận tương quan\ncorr_matrix = train_numeric.corr()\n\n# Vẽ biểu đồ heatmap\nplt.figure(figsize=(40, 40))\nsns.heatmap(corr_matrix, annot=True, cmap='YlGnBu', linewidths=0.5, fmt='.2f', vmin=-1, vmax=1)\nplt.title('Ma Trận Tương Quan', fontsize=24, weight='bold', color='black')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T11:58:04.037349Z","iopub.execute_input":"2024-12-22T11:58:04.037698Z","iopub.status.idle":"2024-12-22T11:58:14.871578Z","shell.execute_reply.started":"2024-12-22T11:58:04.037661Z","shell.execute_reply":"2024-12-22T11:58:14.869463Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n\ndef feature_engineering(df):\n    # season_cols = [col for col in df.columns if 'Season' in col]\n    # df = df.drop(season_cols, axis=1) \n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    \n    return df\n\nclass AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim * 3), nn.ReLU(),\n            nn.Linear(encoding_dim * 3, encoding_dim * 2), nn.ReLU(),\n            nn.Linear(encoding_dim * 2, encoding_dim), nn.ReLU()\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim * 2), nn.ReLU(),\n            nn.Linear(input_dim * 2, input_dim * 3), nn.ReLU(),\n            nn.Linear(input_dim * 3, input_dim), nn.Sigmoid()\n        )\n\n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded\n\n\n\n\ndef perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    \n    data_tensor = torch.FloatTensor(df_scaled)\n    \n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n    \n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(autoencoder.parameters())\n    \n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n            \n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n                 \n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n        \n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n    \n    return df_encoded","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T11:58:14.873983Z","iopub.execute_input":"2024-12-22T11:58:14.874506Z","iopub.status.idle":"2024-12-22T11:58:14.898064Z","shell.execute_reply.started":"2024-12-22T11:58:14.874455Z","shell.execute_reply":"2024-12-22T11:58:14.896533Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T11:58:14.899892Z","iopub.execute_input":"2024-12-22T11:58:14.900435Z","iopub.status.idle":"2024-12-22T11:58:14.959598Z","shell.execute_reply.started":"2024-12-22T11:58:14.900380Z","shell.execute_reply":"2024-12-22T11:58:14.958456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T11:58:14.961129Z","iopub.execute_input":"2024-12-22T11:58:14.961566Z","iopub.status.idle":"2024-12-22T11:58:14.975540Z","shell.execute_reply.started":"2024-12-22T11:58:14.961519Z","shell.execute_reply":"2024-12-22T11:58:14.974398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Xử lý parquet thành các đặc trưng mới\ntrain_ts_0 = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts_0 = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T11:58:14.976882Z","iopub.execute_input":"2024-12-22T11:58:14.977231Z","iopub.status.idle":"2024-12-22T11:59:43.825013Z","shell.execute_reply.started":"2024-12-22T11:58:14.977199Z","shell.execute_reply":"2024-12-22T11:59:43.823580Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# test_ts.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T11:59:43.826838Z","iopub.execute_input":"2024-12-22T11:59:43.827207Z","iopub.status.idle":"2024-12-22T11:59:43.832119Z","shell.execute_reply.started":"2024-12-22T11:59:43.827173Z","shell.execute_reply":"2024-12-22T11:59:43.830943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from keras.optimizers import Adam\nfrom sklearn.preprocessing import StandardScaler\nimport torch.optim as optim\n\n# Xử lý dữ liệu chuỗi thời gian\n\ntrain_ts = train_ts_0.copy()\ntest_ts = test_ts_0.copy()\ndf_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\n\ntrain_ts_encoded = perform_autoencoder(df_train, encoding_dim=60, epochs=100, batch_size=32)\ntest_ts_encoded = perform_autoencoder(df_test, encoding_dim=60, epochs=100, batch_size=32)\n\ntime_series_cols = train_ts_encoded.columns.tolist()\ntrain_ts_encoded[\"id\"]=train_ts[\"id\"]\ntest_ts_encoded['id']=test_ts[\"id\"]\n\n# Merge dữ liệu chuỗi thời gian vào train và test\ntrain = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')\n\ntest['sii'] = np.nan\ntest['PCIAT-PCIAT_Total'] = np.nan\n\ntrain = feature_engineering(train)\ntrain = train.dropna(thresh=10, axis=0)\ntest = feature_engineering(test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T11:59:43.833615Z","iopub.execute_input":"2024-12-22T11:59:43.834095Z","iopub.status.idle":"2024-12-22T12:00:01.453931Z","shell.execute_reply.started":"2024-12-22T11:59:43.834019Z","shell.execute_reply":"2024-12-22T12:00:01.451922Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:01.457213Z","iopub.execute_input":"2024-12-22T12:00:01.458142Z","iopub.status.idle":"2024-12-22T12:00:01.486440Z","shell.execute_reply.started":"2024-12-22T12:00:01.458093Z","shell.execute_reply":"2024-12-22T12:00:01.485124Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# feature selection\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW', 'PCIAT-PCIAT_Total']\n\nfeaturesCols += time_series_cols\nfeaturesCols += season_cols\n\ntrain = train.dropna(subset='sii')\n\n# Loại bỏ các cột chứa >= 30% giá trị bị thiếu\ncols_to_remove = [col for col in featuresCols if train[col].isnull().mean() >= 0.3]\nfeaturesCols = [col for col in featuresCols if col not in cols_to_remove]\n\ntrain = train[featuresCols]\ntest = test[featuresCols] # Đến bước này, train và test đang có các cột giống nhau\nseason_cols = [col for col in featuresCols if 'Season' in col] # List các cột season được giữ lại\n\n# điền giá trị thiếu cho train data\nimputer = KNNImputer(n_neighbors=5)\nnumeric_cols = train.select_dtypes(include=['float64', 'int64']).columns\nimputed_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\n\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\n\ntrain = train_imputed\n\n# Điền giá trị thiếu cho các cột season trong train data\ncat_cols = train.select_dtypes(include=['object']).columns\ncat_imputer = SimpleImputer(strategy='most_frequent')\ntrain[cat_cols] = cat_imputer.fit_transform(train[cat_cols])\n\n# Chuyển giá trị season sang số nguyên\nseason_mapping = {'Spring': 1, 'Summer': 2, 'Fall': 3, 'Winter': 4}\nfor col in season_cols:\n    train[col] = train[col].map(season_mapping)\n\n# imputation cho test data (transform từ train data)\ntest[numeric_cols] = imputer.transform(test[numeric_cols])\ntest[cat_cols] = cat_imputer.transform(test[cat_cols])\n\n# Chuyển giá trị season sang số nguyên\nseason_mapping = {'Spring': 1, 'Summer': 2, 'Fall': 3, 'Winter': 4}\nfor col in season_cols:\n    test[col] = test[col].map(season_mapping)\n\ntest = test.drop('sii', axis=1)\ntest = test.drop('PCIAT-PCIAT_Total', axis=1) # Đến đây, train hơn test 2 cột 'sii' và 'PCIAT-PCIAT_TOTAL'\n\ntrain.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:01.488556Z","iopub.execute_input":"2024-12-22T12:00:01.489089Z","iopub.status.idle":"2024-12-22T12:00:02.161662Z","shell.execute_reply.started":"2024-12-22T12:00:01.489016Z","shell.execute_reply":"2024-12-22T12:00:02.157973Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:02.163468Z","iopub.execute_input":"2024-12-22T12:00:02.163950Z","iopub.status.idle":"2024-12-22T12:00:02.189903Z","shell.execute_reply.started":"2024-12-22T12:00:02.163898Z","shell.execute_reply":"2024-12-22T12:00:02.187955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:02.191977Z","iopub.execute_input":"2024-12-22T12:00:02.192677Z","iopub.status.idle":"2024-12-22T12:00:02.206456Z","shell.execute_reply.started":"2024-12-22T12:00:02.192596Z","shell.execute_reply":"2024-12-22T12:00:02.201354Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:02.209040Z","iopub.execute_input":"2024-12-22T12:00:02.210163Z","iopub.status.idle":"2024-12-22T12:00:02.222594Z","shell.execute_reply.started":"2024-12-22T12:00:02.210108Z","shell.execute_reply":"2024-12-22T12:00:02.220790Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:02.224266Z","iopub.execute_input":"2024-12-22T12:00:02.224748Z","iopub.status.idle":"2024-12-22T12:00:02.235575Z","shell.execute_reply.started":"2024-12-22T12:00:02.224697Z","shell.execute_reply":"2024-12-22T12:00:02.234153Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:02.243208Z","iopub.execute_input":"2024-12-22T12:00:02.243652Z","iopub.status.idle":"2024-12-22T12:00:02.286953Z","shell.execute_reply.started":"2024-12-22T12:00:02.243611Z","shell.execute_reply":"2024-12-22T12:00:02.285522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:02.288624Z","iopub.execute_input":"2024-12-22T12:00:02.289135Z","iopub.status.idle":"2024-12-22T12:00:02.325799Z","shell.execute_reply.started":"2024-12-22T12:00:02.289084Z","shell.execute_reply":"2024-12-22T12:00:02.324667Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:02.327332Z","iopub.execute_input":"2024-12-22T12:00:02.327693Z","iopub.status.idle":"2024-12-22T12:00:02.339418Z","shell.execute_reply.started":"2024-12-22T12:00:02.327660Z","shell.execute_reply":"2024-12-22T12:00:02.338110Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.columns.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:02.340827Z","iopub.execute_input":"2024-12-22T12:00:02.341214Z","iopub.status.idle":"2024-12-22T12:00:02.354148Z","shell.execute_reply.started":"2024-12-22T12:00:02.341179Z","shell.execute_reply":"2024-12-22T12:00:02.352730Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.columns.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:02.355714Z","iopub.execute_input":"2024-12-22T12:00:02.356119Z","iopub.status.idle":"2024-12-22T12:00:02.373784Z","shell.execute_reply.started":"2024-12-22T12:00:02.356042Z","shell.execute_reply":"2024-12-22T12:00:02.372377Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.dropna(subset = [\"sii\"], inplace = True)\ntrain.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:02.375375Z","iopub.execute_input":"2024-12-22T12:00:02.375757Z","iopub.status.idle":"2024-12-22T12:00:02.401430Z","shell.execute_reply.started":"2024-12-22T12:00:02.375713Z","shell.execute_reply":"2024-12-22T12:00:02.400158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"np.random.seed(42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:02.403268Z","iopub.execute_input":"2024-12-22T12:00:02.403648Z","iopub.status.idle":"2024-12-22T12:00:02.411163Z","shell.execute_reply.started":"2024-12-22T12:00:02.403614Z","shell.execute_reply":"2024-12-22T12:00:02.409711Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:02.412830Z","iopub.execute_input":"2024-12-22T12:00:02.413757Z","iopub.status.idle":"2024-12-22T12:00:02.441291Z","shell.execute_reply.started":"2024-12-22T12:00:02.413687Z","shell.execute_reply":"2024-12-22T12:00:02.439980Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:02.442719Z","iopub.execute_input":"2024-12-22T12:00:02.443106Z","iopub.status.idle":"2024-12-22T12:00:02.452868Z","shell.execute_reply.started":"2024-12-22T12:00:02.443034Z","shell.execute_reply":"2024-12-22T12:00:02.451633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:02.454256Z","iopub.execute_input":"2024-12-22T12:00:02.454628Z","iopub.status.idle":"2024-12-22T12:00:02.472164Z","shell.execute_reply.started":"2024-12-22T12:00:02.454585Z","shell.execute_reply":"2024-12-22T12:00:02.470283Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Phân chia dữ liệu","metadata":{}},{"cell_type":"code","source":"# Chuẩn bị dữ liệu huấn luyện\ncom_columns = train.columns.intersection(test.columns).tolist()\nX = train[com_columns]\ny = train['PCIAT-PCIAT_Total']\nX_test_1 = test[com_columns]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:02.473698Z","iopub.execute_input":"2024-12-22T12:00:02.474952Z","iopub.status.idle":"2024-12-22T12:00:02.487998Z","shell.execute_reply.started":"2024-12-22T12:00:02.474902Z","shell.execute_reply":"2024-12-22T12:00:02.486697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Bảng quy đổi PCIAT-PCIAT_TOTAL sang sii\ndef convert(scores):\n    scores = np.array(scores)*1.25\n    pred = np.zeros_like(scores)\n    pred[scores <= 30] = 0\n    pred[(scores > 30) & (scores < 50)] = 1\n    pred[(scores >= 50) & (scores < 80)] = 2\n    pred[scores >= 80] = 3\n    return pred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:02.489534Z","iopub.execute_input":"2024-12-22T12:00:02.489880Z","iopub.status.idle":"2024-12-22T12:00:02.501548Z","shell.execute_reply.started":"2024-12-22T12:00:02.489847Z","shell.execute_reply":"2024-12-22T12:00:02.499844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.model_selection import StratifiedKFold, cross_val_score\nfrom sklearn.metrics import make_scorer, cohen_kappa_score\n\n# Hàm kappa\ndef quadratic_kappa(y_true, y_pred):\n    y_true_cat = convert(y_true)\n    y_pred_cat = convert(y_pred)\n    return cohen_kappa_score(y_true_cat, y_pred_cat, weights = 'quadratic')\n\n# Scorer cho cross_val_score\nkappa_scorer = make_scorer(quadratic_kappa, greater_is_better=True)\n\n# Khởi tạo StratifiedKFold\nskf = StratifiedKFold(n_splits=10)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:02.503216Z","iopub.execute_input":"2024-12-22T12:00:02.503598Z","iopub.status.idle":"2024-12-22T12:00:02.520394Z","shell.execute_reply.started":"2024-12-22T12:00:02.503563Z","shell.execute_reply":"2024-12-22T12:00:02.519115Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Cài đặt tham số","metadata":{}},{"cell_type":"code","source":"# Model parameters for LightGBM\n\nLight_Params = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01,  # Increased from 2.68e-06\n    'device': 'cpu'\n\n}\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': 42,\n    'tree_method': 'hist',\n\n}\n\nXGB_Params1 = {'max_depth': 3, 'n_estimators': 59, 'learning_rate': 0.07327652118259573, 'subsample': 0.5968194045365575, 'colsample_bytree': 0.9123669348125403}\n\n# CatBoost parameters\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': 42,\n    'verbose': 0,\n    'l2_leaf_reg': 15,  # Increase from 10\n    'task_type': 'CPU'\n\n}\n\n# Tham số cho RF\nRF_Params = {\n    'n_estimators': 200,\n    'max_depth': 6,\n    'max_features': 0.8,\n    'min_samples_split': 2,\n    'min_samples_leaf': 1,\n    'bootstrap': True,\n    'random_state': SEED\n}\n\n# Tham số cho GB\nHistGB_Params = {\n    'max_iter': 200, \n    'max_depth': 6,  \n    'learning_rate': 0.5, \n    'loss': 'squared_error', \n    'max_leaf_nodes': None,  \n    'random_state': SEED\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:02.521883Z","iopub.execute_input":"2024-12-22T12:00:02.522341Z","iopub.status.idle":"2024-12-22T12:00:02.533760Z","shell.execute_reply.started":"2024-12-22T12:00:02.522291Z","shell.execute_reply":"2024-12-22T12:00:02.532486Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model 1","metadata":{}},{"cell_type":"markdown","source":"có time_series, có season, PCIAT-PCIAT_Total -> sii\nlgbm, xgb, catboost -> voting regressor","metadata":{}},{"cell_type":"code","source":"# Create model instances\nLight = LGBMRegressor(**Light_Params, random_state = 42, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params1)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n# Bước 2: Tạo Voting Regressor\nvoting_regressor = VotingRegressor(\n    estimators=[\n        ('lightgbm', Light),\n        ('xgboost', XGB_Model),\n        ('catboost', CatBoost_Model)\n    ],\n    weights=[4, 4, 5]  # Đặt trọng số cho từng mô hình\n)\n\n# Bước 3: Huấn luyện Voting Regressor\nvoting_regressor.fit(X, y)\n\n# Bước 4: Dự đoán trên tập test\nX_test_1 = test[com_columns]\nensemble_pred = voting_regressor.predict(X_test_1)\n\n# Bước 5: Chuyển đổi dự đoán\ny_test_1 = convert(ensemble_pred)\ny_test_1 = np.round(y_test_1).astype(int)\n\n# Bước 6: Tạo file submission\nsubmission1 = sub_sample.copy()\nsubmission1['sii'] = y_test_1\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:02.535770Z","iopub.execute_input":"2024-12-22T12:00:02.536265Z","iopub.status.idle":"2024-12-22T12:00:06.533495Z","shell.execute_reply.started":"2024-12-22T12:00:02.536210Z","shell.execute_reply":"2024-12-22T12:00:06.532398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:06.534942Z","iopub.execute_input":"2024-12-22T12:00:06.535347Z","iopub.status.idle":"2024-12-22T12:00:06.546564Z","shell.execute_reply.started":"2024-12-22T12:00:06.535310Z","shell.execute_reply":"2024-12-22T12:00:06.545525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# model_1 = xgb.XGBRegressor(**XGB_Params1)\n# # Đánh giá mô hình\n# scores = cross_val_score(model_1, X, y, cv=skf, scoring=kappa_scorer)\n# print(\"Đánh giá mô hình XGBoost\")\n# print(\"QWK Scores:\", scores)\n# print(\"Mean QWK Score:\", np.mean(scores))\n\n\n# model_2 = cb.CatBoostRegressor(**CatBoost_Params)\n# # Đánh giá mô hình\n# scores = cross_val_score(model_2, X, y, cv=skf, scoring=kappa_scorer)\n# print(\"Đánh giá mô hình CatBoost\")\n# print(\"QWK Scores:\", scores)\n# print(\"Mean QWK Score:\", np.mean(scores))\n\n# # model_3 = lgbm.LGBMRegressor(**Light_Params)\n# #  # Đánh giá mô hình\n# # scores = cross_val_score(model_3, X, y, cv=skf, scoring=kappa_scorer)\n# # print(\"Đánh giá mô hình LGBM\")\n# # print(\"QWK Scores:\", scores)\n# # print(\"Mean QWK Score:\", np.mean(scores))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:06.548150Z","iopub.execute_input":"2024-12-22T12:00:06.548519Z","iopub.status.idle":"2024-12-22T12:00:06.559775Z","shell.execute_reply.started":"2024-12-22T12:00:06.548485Z","shell.execute_reply":"2024-12-22T12:00:06.558598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model 2","metadata":{}},{"cell_type":"markdown","source":"tương tự 1, nhưng ko time-series, có season, chuyển đổi PCIAT-PCIAT_Total -> sii,\n\nthêm stacking cho ensemble","metadata":{}},{"cell_type":"code","source":"# # Đọc dữ liệu\n# train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n# test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n# sample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n# season_cols = [col for col in test.columns if 'Season' in col]\n\n# test['sii'] = np.nan\n# test['PCIAT-PCIAT_Total'] = np.nan\n\n# train = feature_engineering(train)\n# train = train.dropna(thresh=10, axis=0)\n# test = feature_engineering(test)\n\n# # feature selection\n# featuresCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n#                 'CGAS-CGAS_Score', 'Physical-BMI',\n#                 'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n#                 'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n#                 'Fitness_Endurance-Max_Stage',\n#                 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n#                 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n#                 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n#                 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n#                 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n#                 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n#                 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n#                 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n#                 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n#                 'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n#                 'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n#                 'SDS-SDS_Total_T',\n#                 'PreInt_EduHx-computerinternet_hoursday', 'sii', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n#                 'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n#                 'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW', 'PCIAT-PCIAT_Total']\n\n# featuresCols += season_cols\n\n# train = train.dropna(subset='sii')\n\n# # Loại bỏ các cột chứa >= 30% giá trị bị thiếu\n# cols_to_remove = [col for col in featuresCols if train[col].isnull().mean() >= 0.3]\n# featuresCols = [col for col in featuresCols if col not in cols_to_remove]\n\n# train = train[featuresCols]\n# test = test[featuresCols] # Đến bước này, train và test đang có các cột giống nhau\n# season_cols = [col for col in featuresCols if 'Season' in col] # List các cột season được giữ lại\n\n# # điền giá trị thiếu cho train data\n# imputer = KNNImputer(n_neighbors=5)\n# numeric_cols = train.select_dtypes(include=['float64', 'int64']).columns\n# imputed_data = imputer.fit_transform(train[numeric_cols])\n# train_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\n# train_imputed['sii'] = train_imputed['sii'].round().astype(int)\n\n# for col in train.columns:\n#     if col not in numeric_cols:\n#         train_imputed[col] = train[col]\n\n# train = train_imputed\n\n# # Điền giá trị thiếu cho các cột season trong train data\n# cat_cols = train.select_dtypes(include=['object']).columns\n# cat_imputer = SimpleImputer(strategy='most_frequent')\n# train[cat_cols] = cat_imputer.fit_transform(train[cat_cols])\n\n# # Chuyển giá trị season sang số nguyên\n# season_mapping = {'Spring': 1, 'Summer': 2, 'Fall': 3, 'Winter': 4}\n# for col in season_cols:\n#     train[col] = train[col].map(season_mapping)\n\n# # imputation cho test data (transform từ train data)\n# test[numeric_cols] = imputer.transform(test[numeric_cols])\n# test[cat_cols] = cat_imputer.transform(test[cat_cols])\n\n# # Chuyển giá trị season sang số nguyên\n# season_mapping = {'Spring': 1, 'Summer': 2, 'Fall': 3, 'Winter': 4}\n# for col in season_cols:\n#     test[col] = test[col].map(season_mapping)\n\n# test = test.drop('sii', axis=1)\n# test = test.drop('PCIAT-PCIAT_Total', axis=1) # Đến đây, train hơn test 2 cột 'sii' và 'PCIAT-PCIAT_TOTAL'\n\n# train.isna().sum()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:06.561409Z","iopub.execute_input":"2024-12-22T12:00:06.561804Z","iopub.status.idle":"2024-12-22T12:00:06.575025Z","shell.execute_reply.started":"2024-12-22T12:00:06.561769Z","shell.execute_reply":"2024-12-22T12:00:06.573791Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Đặt SEED\n# SEED = 42\n\n# # Bước 1: Khởi tạo các mô hình cơ bản với random_state=SEED\n# Light = LGBMRegressor(**Light_Params, random_state=SEED, verbose=-1, n_estimators=300)\n# XGB_Model = XGBRegressor(**XGB_Params1, random_state=SEED)\n# CatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n# # Bước 2: Voting Regressor\n# voting_regressor = VotingRegressor(\n#     estimators=[\n#         ('lightgbm', Light),\n#         ('xgboost', XGB_Model),\n#         ('catboost', CatBoost_Model)\n#     ],\n#     weights=[4, 4, 5]  # Đặt trọng số cho từng mô hình\n# )\n\n# # Bước 3: Stacking Regressor\n# stacking_regressor = StackingRegressor(\n#     estimators=[\n#         ('lightgbm', Light),\n#         ('xgboost', XGB_Model),\n#         ('catboost', CatBoost_Model)\n#     ],\n#     final_estimator=Ridge(alpha=1.0, random_state=SEED),  # Meta-learner với random_state\n#     cv=5  # Cross-validation\n# )\n\n# # Bước 4: Kết hợp cả hai ensemble vào mô hình Stacking Regressor cuối cùng\n# final_ensemble = StackingRegressor(\n#     estimators=[\n#         ('voting', voting_regressor),\n#         ('stacking', stacking_regressor)\n#     ],\n#     final_estimator=Ridge(alpha=0.5, random_state=SEED),  # Meta-learner cuối cùng\n#     cv=5  # Cross-validation\n# )\n\n# # Bước 5: Huấn luyện mô hình cuối cùng\n# final_ensemble.fit(X, y)\n\n# # Bước 6: Dự đoán trên tập test\n# final_pred = final_ensemble.predict(X_test_1)\n\n# # Bước 7: Chuyển đổi dự đoán\n# y_test_final = convert(final_pred)\n# y_test_final = np.round(y_test_final).astype(int)\n\n# # Bước 8: Tạo file submission\n# submission_2 = sub_sample.copy()\n# submission_2['sii'] = y_test_final\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:06.576548Z","iopub.execute_input":"2024-12-22T12:00:06.576916Z","iopub.status.idle":"2024-12-22T12:00:06.596651Z","shell.execute_reply.started":"2024-12-22T12:00:06.576883Z","shell.execute_reply":"2024-12-22T12:00:06.595426Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission_2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:06.600591Z","iopub.execute_input":"2024-12-22T12:00:06.601019Z","iopub.status.idle":"2024-12-22T12:00:06.612367Z","shell.execute_reply.started":"2024-12-22T12:00:06.600979Z","shell.execute_reply":"2024-12-22T12:00:06.611212Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission_2.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:06.613794Z","iopub.execute_input":"2024-12-22T12:00:06.614207Z","iopub.status.idle":"2024-12-22T12:00:06.623300Z","shell.execute_reply.started":"2024-12-22T12:00:06.614171Z","shell.execute_reply":"2024-12-22T12:00:06.622127Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model 3","metadata":{}},{"cell_type":"markdown","source":"Có time_series, ko season, dự đoán trực tiếp sii\nThêm Random Forest, Gradient Boost","metadata":{}},{"cell_type":"code","source":"# Đọc dữ liệu\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsub_sample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n# Xử lý parquet thành các đặc trưng mới\ntrain_ts = train_ts_0.copy()\ntest_ts = test_ts_0.copy()\n\nfrom keras.optimizers import Adam\nfrom sklearn.preprocessing import StandardScaler\nimport torch.optim as optim\n\n# Xử lý dữ liệu chuỗi thời gian\ndf_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\n\ntrain_ts_encoded = perform_autoencoder(df_train, encoding_dim=60, epochs=100, batch_size=32)\ntest_ts_encoded = perform_autoencoder(df_test, encoding_dim=60, epochs=100, batch_size=32)\n\ntime_series_cols = train_ts_encoded.columns.tolist()\ntrain_ts_encoded[\"id\"]=train_ts[\"id\"]\ntest_ts_encoded['id']=test_ts[\"id\"]\n\n# Merge dữ liệu chuỗi thời gian vào train và test\ntrain = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')\n\ntest['sii'] = np.nan\ntest['PCIAT-PCIAT_Total'] = np.nan\n\ntrain = feature_engineering(train)\ntrain = train.dropna(thresh=10, axis=0)\ntest = feature_engineering(test)\n\n# feature selection\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW']\n\nfeaturesCols += time_series_cols\n\ntrain = train.dropna(subset='sii')\n\n# Loại bỏ các cột chứa >= 30% giá trị bị thiếu\ncols_to_remove = [col for col in featuresCols if train[col].isnull().mean() >= 0.3]\nfeaturesCols = [col for col in featuresCols if col not in cols_to_remove]\n\ntrain = train[featuresCols]\ntest = test[featuresCols] # Đến bước này, train và test đang có các cột giống nhau\n\n# điền giá trị thiếu cho train data\nimputer = KNNImputer(n_neighbors=5)\nnumeric_cols = train.select_dtypes(include=['float64', 'int64']).columns\nimputed_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\n\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\n\ntrain = train_imputed\n\n# imputation cho test data (transform từ train data)\ntest[numeric_cols] = imputer.transform(test[numeric_cols])\n\ntest = test.drop('sii', axis=1) # Đến đây, train hơn test 1 cột là 'sii'\n\ntrain.isna().sum()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:06.624906Z","iopub.execute_input":"2024-12-22T12:00:06.625674Z","iopub.status.idle":"2024-12-22T12:00:20.305131Z","shell.execute_reply.started":"2024-12-22T12:00:06.625636Z","shell.execute_reply":"2024-12-22T12:00:20.302279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:20.308094Z","iopub.execute_input":"2024-12-22T12:00:20.308783Z","iopub.status.idle":"2024-12-22T12:00:20.335420Z","shell.execute_reply.started":"2024-12-22T12:00:20.308714Z","shell.execute_reply":"2024-12-22T12:00:20.331404Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Chuẩn bị dữ liệu huấn luyện\ncom_columns = train.columns.intersection(test.columns).tolist()\nX = train[com_columns]\ny = train[\"sii\"]\nX_test_1 = test[com_columns]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:20.337468Z","iopub.execute_input":"2024-12-22T12:00:20.337948Z","iopub.status.idle":"2024-12-22T12:00:20.356788Z","shell.execute_reply.started":"2024-12-22T12:00:20.337898Z","shell.execute_reply":"2024-12-22T12:00:20.355068Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Đặt SEED\nSEED = 42\n\n# Bước 1: Khởi tạo các mô hình với tham số đã tối ưu\nLight = LGBMRegressor(**Light_Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params1, random_state=SEED)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\nRF_Model = RandomForestRegressor(**RF_Params)\nGB_Model = HistGradientBoostingRegressor(**HistGB_Params)\n\n\n# Bước 2: Tạo Stacking Regressor\nstacking_regressor = StackingRegressor(\n    estimators=[\n        ('lightgbm', Light),\n        ('xgboost', XGB_Model),\n        ('catboost', CatBoost_Model),\n        ('random_forest', RF_Model),\n        ('hist_gb', GB_Model)\n    ],\n    final_estimator=Ridge(alpha=1.0, random_state=SEED),  # Meta-learner\n    cv=5  # Cross-validation\n)\n\n# Bước 3: Huấn luyện Stacking Regressor\nstacking_regressor.fit(X, y)\n\n# Bước 4: Dự đoán trên tập test\nstacking_pred = stacking_regressor.predict(X_test_1)\n\n# Bước 5: Làm tròn\ny_test_final = np.round(stacking_pred).astype(int)\n\n# Bước 6: Tạo submission\nsubmission_3 = sub_sample.copy()\nsubmission_3['sii'] = y_test_final\nsubmission_3.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:20.360662Z","iopub.execute_input":"2024-12-22T12:00:20.362096Z","iopub.status.idle":"2024-12-22T12:00:41.463011Z","shell.execute_reply.started":"2024-12-22T12:00:20.362003Z","shell.execute_reply":"2024-12-22T12:00:41.461370Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_3","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:41.464671Z","iopub.execute_input":"2024-12-22T12:00:41.465124Z","iopub.status.idle":"2024-12-22T12:00:41.486087Z","shell.execute_reply.started":"2024-12-22T12:00:41.465077Z","shell.execute_reply":"2024-12-22T12:00:41.483837Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Result","metadata":{}},{"cell_type":"code","source":"# # Bỏ phiếu đa số\n# submission1 = submission1.sort_values(by='id').reset_index(drop=True)\n# submission2 = submission_2.sort_values(by='id').reset_index(drop=True)\n# submission3 = submission_3.sort_values(by='id').reset_index(drop=True)\n\n# combined = pd.DataFrame({\n#     'id': submission1['id'],\n#     'sii_1': submission1['sii'],\n#     'sii_2': submission2['sii'],\n#     'sii_3': submission3['sii']\n# })\n\n# def majority_vote(row):\n#    return row.mode()[0]\n\n# combined['final_sii'] = combined[['sii_1', 'sii_2', 'sii_3']].apply(majority_vote, axis=1)\n\n# final_submission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\n# final_submission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:41.488266Z","iopub.execute_input":"2024-12-22T12:00:41.490923Z","iopub.status.idle":"2024-12-22T12:00:41.506034Z","shell.execute_reply.started":"2024-12-22T12:00:41.490858Z","shell.execute_reply":"2024-12-22T12:00:41.504688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# final_submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:00:41.507819Z","iopub.execute_input":"2024-12-22T12:00:41.508286Z","iopub.status.idle":"2024-12-22T12:00:41.522367Z","shell.execute_reply.started":"2024-12-22T12:00:41.508238Z","shell.execute_reply":"2024-12-22T12:00:41.521315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}