{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":7453542,"sourceType":"datasetVersion","datasetId":921302}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip -q install /kaggle/input/pytorchtabnet/pytorch_tabnet-4.1.0-py3-none-any.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:51:14.958147Z","iopub.execute_input":"2024-12-22T05:51:14.958943Z","iopub.status.idle":"2024-12-22T05:51:55.121589Z","shell.execute_reply.started":"2024-12-22T05:51:14.958906Z","shell.execute_reply":"2024-12-22T05:51:55.120476Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Import thư viện","metadata":{}},{"cell_type":"code","source":"import os\nimport re\nimport warnings\nimport random\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom concurrent.futures import ThreadPoolExecutor\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom scipy.stats import kurtosis\n\nfrom sklearn.base import clone, BaseEstimator, RegressorMixin\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import SimpleImputer, IterativeImputer, KNNImputer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.utils import resample\nfrom scipy.optimize import minimize\n\n# Gradient Boosting Models\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\n\nfrom pytorch_tabnet.tab_model import TabNetRegressor\nfrom pytorch_tabnet.callbacks import Callback\n\nfrom imblearn.over_sampling import SMOTE\n\nwarnings.filterwarnings('ignore')\n# Không giới hạn số cột với pandas\npd.options.display.max_columns = None\n\nSEED = 42\nN_SPLITS = 5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:51:55.123453Z","iopub.execute_input":"2024-12-22T05:51:55.123741Z","iopub.status.idle":"2024-12-22T05:52:00.303167Z","shell.execute_reply.started":"2024-12-22T05:51:55.123714Z","shell.execute_reply":"2024-12-22T05:52:00.302345Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Solution 1","metadata":{}},{"cell_type":"markdown","source":"## Load csv data","metadata":{}},{"cell_type":"code","source":"train_csv = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_csv = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample_sub = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:52:00.304261Z","iopub.execute_input":"2024-12-22T05:52:00.305027Z","iopub.status.idle":"2024-12-22T05:52:00.372153Z","shell.execute_reply.started":"2024-12-22T05:52:00.304986Z","shell.execute_reply":"2024-12-22T05:52:00.371389Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_csv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:52:00.374550Z","iopub.execute_input":"2024-12-22T05:52:00.375080Z","iopub.status.idle":"2024-12-22T05:52:00.467025Z","shell.execute_reply.started":"2024-12-22T05:52:00.375036Z","shell.execute_reply":"2024-12-22T05:52:00.466211Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_csv.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:52:00.468254Z","iopub.execute_input":"2024-12-22T05:52:00.468823Z","iopub.status.idle":"2024-12-22T05:52:00.474198Z","shell.execute_reply.started":"2024-12-22T05:52:00.468768Z","shell.execute_reply":"2024-12-22T05:52:00.473262Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load time series data","metadata":{}},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n\n    df['time_of_day'] = pd.to_datetime(df['time_of_day'])\n    df.set_index('time_of_day', inplace=True)\n\n    # Daily aggregation\n    daily_features = df.resample('D').agg({\n        'X': ['mean', 'std', 'max', 'min', 'median', 'skew', lambda x: kurtosis(x, nan_policy='omit')],\n        'Y': ['mean', 'std', 'max', 'min', 'median', 'skew', lambda x: kurtosis(x, nan_policy='omit')],\n        'Z': ['mean', 'std', 'max', 'min', 'median', 'skew', lambda x: kurtosis(x, nan_policy='omit')],\n        'enmo': ['mean', 'std', 'max', 'min', 'median', 'skew', lambda x: kurtosis(x, nan_policy='omit')],\n        'light': ['mean', 'std'],\n        'battery_voltage': ['mean', 'std'],\n        'non-wear_flag': 'sum'\n    })\n\n    # Flatten the MultiIndex columns\n    daily_features.columns = ['_'.join(col).strip() for col in daily_features.columns.values]\n    \n    # Add additional features\n    daily_features['activity_count'] = (df['enmo'] > 0.1).resample('D').sum() \n    daily_features['days_active'] = (df['non-wear_flag'] == 0).resample('D').sum() \n    \n    # Rate of change features\n    daily_features['enmo_change'] = daily_features['enmo_mean'].diff()\n    \n    # Rolling statistics\n    rolling_windows = [5, 10, 15]  # Days\n    for window in rolling_windows:\n        daily_features[f'enmo_rolling_mean_{window}'] = daily_features['enmo_mean'].rolling(window=window).mean()\n        daily_features[f'enmo_rolling_std_{window}'] = daily_features['enmo_std'].rolling(window=window).std()\n\n    # Frequency domain features using FFT\n    for axis in ['X', 'Y', 'Z', 'enmo']:\n        freq_features = np.fft.fft(df[axis])\n        daily_features[f'{axis}_dominant_freq'] = np.abs(freq_features).argmax()\n\n    # Reset index to retain 'id'\n    daily_features.reset_index(inplace=True)\n    \n    # Extract child ID from filename\n    child_id = filename.split('=')[1]\n    daily_features['id'] = child_id\n\n    return daily_features\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    # Concatenate all results into a single DataFrame\n    features_df = pd.concat(results, ignore_index=True)\n\n    return features_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:52:00.475512Z","iopub.execute_input":"2024-12-22T05:52:00.475966Z","iopub.status.idle":"2024-12-22T05:52:00.489264Z","shell.execute_reply.started":"2024-12-22T05:52:00.475924Z","shell.execute_reply":"2024-12-22T05:52:00.488457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ts_train = load_time_series('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet')\nts_test = load_time_series('/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:52:00.490268Z","iopub.execute_input":"2024-12-22T05:52:00.490590Z","iopub.status.idle":"2024-12-22T05:57:19.347128Z","shell.execute_reply.started":"2024-12-22T05:52:00.490557Z","shell.execute_reply":"2024-12-22T05:57:19.346277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"columns_to_drop = [\n    'enmo_change',\n    'enmo_rolling_mean_5',\n    'enmo_rolling_std_5',\n    'enmo_rolling_mean_10',\n    'enmo_rolling_std_10',\n    'enmo_rolling_mean_15',\n    'enmo_rolling_std_15',\n    'time_of_day'\n] \n\nts_train.drop(columns=columns_to_drop, inplace=True)\nts_test.drop(columns=columns_to_drop, inplace=True)\n\nts_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:19.348056Z","iopub.execute_input":"2024-12-22T05:57:19.348289Z","iopub.status.idle":"2024-12-22T05:57:19.384884Z","shell.execute_reply.started":"2024-12-22T05:57:19.348266Z","shell.execute_reply":"2024-12-22T05:57:19.383918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train_ts = ts_train.drop('id', axis=1)\ndf_test_ts = ts_test.drop('id', axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:19.385931Z","iopub.execute_input":"2024-12-22T05:57:19.386212Z","iopub.status.idle":"2024-12-22T05:57:19.394446Z","shell.execute_reply.started":"2024-12-22T05:57:19.386185Z","shell.execute_reply":"2024-12-22T05:57:19.393727Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Encode dữ liệu ts","metadata":{}},{"cell_type":"code","source":"class AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim*3),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*3, encoding_dim*2),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*2, encoding_dim),\n            nn.ReLU()\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim*2),\n            nn.ReLU(),\n            nn.Linear(input_dim*2, input_dim*3),\n            nn.ReLU(),\n            nn.Linear(input_dim*3, input_dim),\n            nn.Sigmoid()\n        )\n        \n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded\n\n\ndef perform_autoencoder(df, encoding_dim=60, epochs=200, batch_size=32):\n    scaler = StandardScaler()\n    # Exclude non-numeric columns\n    df_numeric = df.select_dtypes(include=[np.number])\n    \n    # Check for and handle NaN values\n    if df_numeric.isnull().values.any():\n        df_numeric.fillna(df_numeric.mean(), inplace=True)\n\n    if np.isinf(df_numeric).values.any():\n        df_numeric.replace([np.inf, -np.inf], np.nan, inplace=True)\n        df_numeric.fillna(df_numeric.mean(), inplace=True)\n    df_scaled = scaler.fit_transform(df_numeric)\n    \n    data_tensor = torch.FloatTensor(df_scaled)\n    \n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n    \n    criterion = nn.MSELoss()\n    optimizer = optim.AdamW(autoencoder.parameters())\n    \n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n            \n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n                 \n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n        \n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n    \n    return df_encoded","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:19.397330Z","iopub.execute_input":"2024-12-22T05:57:19.397669Z","iopub.status.idle":"2024-12-22T05:57:19.410356Z","shell.execute_reply.started":"2024-12-22T05:57:19.397643Z","shell.execute_reply":"2024-12-22T05:57:19.409637Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ts_train_encoded = perform_autoencoder(df_train_ts)\nts_test_encoded = perform_autoencoder(df_test_ts)\n\nenc_time_series_cols = ts_train_encoded.columns.tolist()\nts_train_encoded[\"id\"]=ts_train[\"id\"]\nts_test_encoded[\"id\"]=ts_test[\"id\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:19.411220Z","iopub.execute_input":"2024-12-22T05:57:19.411471Z","iopub.status.idle":"2024-12-22T05:57:36.236962Z","shell.execute_reply.started":"2024-12-22T05:57:19.411430Z","shell.execute_reply":"2024-12-22T05:57:36.236055Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Xử lý dữ liệu (Có thể cải tiến thêm)","metadata":{}},{"cell_type":"code","source":"# Kết hợp dữ liệu time series\ntrain = pd.merge(train_csv, ts_train_encoded, how=\"left\", on='id')\ntest = pd.merge(test_csv, ts_test_encoded, how=\"left\", on='id')\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1) \n\n# # Không kết hợp\n# train = train_csv.drop('id', axis=1)\n# test = test_csv.drop('id', axis=1) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.238124Z","iopub.execute_input":"2024-12-22T05:57:36.238678Z","iopub.status.idle":"2024-12-22T05:57:36.259554Z","shell.execute_reply.started":"2024-12-22T05:57:36.238649Z","shell.execute_reply":"2024-12-22T05:57:36.258728Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.260605Z","iopub.execute_input":"2024-12-22T05:57:36.260931Z","iopub.status.idle":"2024-12-22T05:57:36.266909Z","shell.execute_reply.started":"2024-12-22T05:57:36.260895Z","shell.execute_reply":"2024-12-22T05:57:36.265908Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"featuresCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\n# featuresCols += enc_time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.267987Z","iopub.execute_input":"2024-12-22T05:57:36.268300Z","iopub.status.idle":"2024-12-22T05:57:36.282499Z","shell.execute_reply.started":"2024-12-22T05:57:36.268268Z","shell.execute_reply":"2024-12-22T05:57:36.281550Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineering(df):\n    # Age features\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['HeartRate_Age_Ratio'] = df['Physical-HeartRate'] / df['Basic_Demos-Age']\n    # BMI features\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    df['BMI_PHR'] = df['Physical-BMI'] * df['Physical-HeartRate']\n\n    # Internet hours features\n    hoursDayAdd1hour = df['PreInt_EduHx-computerinternet_hoursday'] + 1\n    df['Internet_Hours_SDS'] = hoursDayAdd1hour * df['SDS-SDS_Total_T']\n    df['Internet_Hours_PAQ'] = hoursDayAdd1hour * df['PAQ_C-PAQ_C_Total']\n    df['Internet_to_Activity_Ratio'] = hoursDayAdd1hour / df['PAQ_C-PAQ_C_Total']\n    df['Internet_Hours_HeartRate_Ratio'] = hoursDayAdd1hour / df['Physical-HeartRate']\n    df['HeartRate_SDS'] = df['Physical-HeartRate'] * df['SDS-SDS_Total_T']\n    df['SDS_CGAS_Ratio'] = df['SDS-SDS_Total_T'] / df['CGAS-CGAS_Score']\n    df['SDS_PAQ_Interaction'] = df['SDS-SDS_Total_T'] * df['PAQ_C-PAQ_C_Total']\n    df['CGAS_PAQ_Interaction'] = df['CGAS-CGAS_Score'] * df['PAQ_C-PAQ_C_Total']\n    df['SDS_HeartRate_Ratio'] = df['SDS-SDS_Total_T'] / df['Physical-HeartRate']\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.283656Z","iopub.execute_input":"2024-12-22T05:57:36.284031Z","iopub.status.idle":"2024-12-22T05:57:36.293901Z","shell.execute_reply.started":"2024-12-22T05:57:36.283980Z","shell.execute_reply":"2024-12-22T05:57:36.293106Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Impute + Feature Engineering","metadata":{}},{"cell_type":"code","source":"sim_imputer = SimpleImputer()\nknn_imputer = KNNImputer(n_neighbors=5)\n\n# numeric_cols = list(train.select_dtypes(include=['float32', 'int32', 'float64', 'int64']).columns)\n# # imputed_data = sim_imputer.fit_transform(train[numeric_cols])\n# imputed_data = knn_imputer.fit_transform(train[numeric_cols])\n# train_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\n# train_imputed['sii'] = train_imputed['sii'].round().astype(int)\n# for col in train.columns:\n#     if col not in numeric_cols:\n#         train_imputed[col] = train[col]       \n# train = train_imputed","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.294799Z","iopub.execute_input":"2024-12-22T05:57:36.295076Z","iopub.status.idle":"2024-12-22T05:57:36.305594Z","shell.execute_reply.started":"2024-12-22T05:57:36.295044Z","shell.execute_reply":"2024-12-22T05:57:36.304628Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Xử lý dữ liệu category","metadata":{}},{"cell_type":"code","source":"category_cols = [col for col in test_csv.columns if 'Season' in col]\n# train.drop(category_cols, axis=1, inplace=True)\n# test.drop(category_cols, axis=1, inplace=True)\n\ndef update(df):\n    for c in category_cols: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n    \nfor col in category_cols:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.306602Z","iopub.execute_input":"2024-12-22T05:57:36.306917Z","iopub.status.idle":"2024-12-22T05:57:36.368224Z","shell.execute_reply.started":"2024-12-22T05:57:36.306876Z","shell.execute_reply":"2024-12-22T05:57:36.367335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Thêm feature sau khi impute\ntrain = feature_engineering(train)\ntest = feature_engineering(test)\n\ntrain = train.dropna(thresh=train.shape[1]//10, axis=0)\n# Xóa các dòng với cột 'sii' = NaN\ntrain = train.dropna(subset='sii', axis=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.369380Z","iopub.execute_input":"2024-12-22T05:57:36.369696Z","iopub.status.idle":"2024-12-22T05:57:36.403874Z","shell.execute_reply.started":"2024-12-22T05:57:36.369669Z","shell.execute_reply":"2024-12-22T05:57:36.403051Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.404941Z","iopub.execute_input":"2024-12-22T05:57:36.405219Z","iopub.status.idle":"2024-12-22T05:57:36.411172Z","shell.execute_reply.started":"2024-12-22T05:57:36.405192Z","shell.execute_reply":"2024-12-22T05:57:36.410047Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Số lượng mẫu với mỗi class","metadata":{}},{"cell_type":"code","source":"def display_class(df):\n    print(\"\\nClass Distribution:\")\n    print(df['sii'].value_counts(normalize=True))\n    \n    print(\"\\nSamples per Class:\")\n    print(df['sii'].value_counts())\ndisplay_class(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.412372Z","iopub.execute_input":"2024-12-22T05:57:36.412667Z","iopub.status.idle":"2024-12-22T05:57:36.426193Z","shell.execute_reply.started":"2024-12-22T05:57:36.412641Z","shell.execute_reply":"2024-12-22T05:57:36.425320Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Xử lý mất cân bằng dữ liệu","metadata":{}},{"cell_type":"code","source":"# SMOTE\ndef upsample_smote(df):\n    y = df['sii']\n    X = df.drop('sii', axis=1)\n    sampling_strategy = {\n        0: len(y[y == 0]), \n        1: len(y[y == 1]), \n        2: int(len(y[y == 1]) * 2/3),  \n        3: len(y[y == 2]) // 2 \n    }\n    \n    smote = SMOTE(\n        sampling_strategy=sampling_strategy, \n        random_state=SEED\n    )\n    \n    X_resampled, y_resampled = smote.fit_resample(X, y)\n    X_resampled['sii'] = y_resampled\n    return X_resampled\n\n# Random upsampling\nupsampled_mild = resample(\n    train[train['sii'] == 1], \n    replace=True,     \n    n_samples=len(train[train['sii'] == 1]), # 730 => 1460\n    random_state=SEED  \n)\nupsampled_moderate = resample(\n    train[train['sii'] == 2], \n    replace=True,     \n    n_samples=len(train[train['sii'] == 2])*2, # 378 => 1134\n    random_state=SEED  \n)\nupsampled_severe = resample(\n    train[train['sii'] == 3], \n    replace=True,     \n    n_samples=len(train[train['sii'] == 3])*9, # 34 => 340\n    random_state=SEED  \n)\ntrain_data = pd.concat([train, upsampled_mild, upsampled_moderate, upsampled_severe])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.427265Z","iopub.execute_input":"2024-12-22T05:57:36.427547Z","iopub.status.idle":"2024-12-22T05:57:36.437339Z","shell.execute_reply.started":"2024-12-22T05:57:36.427520Z","shell.execute_reply":"2024-12-22T05:57:36.436591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Xử lý dữ liệu quá lớn cho SMOTE\n# for column in train_data.columns:\n#     if np.any(np.isinf(train_data[column])):\n#         col_mean = train_data[column][~np.isinf(train_data[column])].mean()\n#         train_data[column] = train_data[column].replace([np.inf, -np.inf], col_mean)\n# train_data = upsample_smote(train)\n\n# if np.any(np.isinf(train)):\n#     train_data = train.replace([np.inf, -np.inf], np.nan)\n\nif np.any(np.isinf(train_data)):\n    train_data = train_data.replace([np.inf, -np.inf], np.nan)\n\ntestCols = list(train_data.columns)\ntestCols.remove('sii')\ntest_data = test[testCols]\n\ntrain_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.438321Z","iopub.execute_input":"2024-12-22T05:57:36.439024Z","iopub.status.idle":"2024-12-22T05:57:36.618161Z","shell.execute_reply.started":"2024-12-22T05:57:36.438981Z","shell.execute_reply":"2024-12-22T05:57:36.616467Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display_class(train_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.618706Z","iopub.status.idle":"2024-12-22T05:57:36.618986Z","shell.execute_reply.started":"2024-12-22T05:57:36.618839Z","shell.execute_reply":"2024-12-22T05:57:36.618854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.620596Z","iopub.status.idle":"2024-12-22T05:57:36.620865Z","shell.execute_reply.started":"2024-12-22T05:57:36.620733Z","shell.execute_reply":"2024-12-22T05:57:36.620748Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Huấn luyện mô hình","metadata":{}},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_rounder(oof_non_rounded, thresholds):\n    \"\"\"Làm tròn giá trị output\"\"\"\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef train_model(model_class, train_data, test_data):\n    \"\"\"Huấn luyện mô hình với cross-validation\"\"\"\n    X = train_data.drop(['sii'], axis=1)\n    y = train_data['sii']\n\n    skf = StratifiedKFold(n_splits=N_SPLITS, shuffle=True, random_state=SEED)\n\n    train_scores = []\n    val_scores = []\n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_predictions = np.zeros((len(test_data), N_SPLITS))\n\n    for fold, (train_idx, val_idx) in enumerate(tqdm(skf.split(X, y), desc=\"Training Folds\", total=N_SPLITS)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[val_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[val_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_scores.append(train_kappa)\n        val_scores.append(val_kappa)\n\n        test_predictions[:, fold] = model.predict(test_data)\n\n        print(f\"Fold {fold+1} | Train QWK: {train_kappa:.4f} | Validation QWK: {val_kappa:.4f}\")\n\n    print(f\"Mean Train QWK: {np.mean(train_scores):.4f}\")\n    print(f\"Mean Validation QWK: {np.mean(val_scores):.4f}\")\n\n    # Tối ưu threshold\n    kappa_optimizer = minimize(\n        evaluate_predictions,\n        x0=[0.5, 1.5, 2.5], \n        args=(y, oof_non_rounded), \n        method='Nelder-Mead'\n    )\n    assert kappa_optimizer.success, \"Optimization didn't converge.\"\n\n    oof_tuned = threshold_rounder(oof_non_rounded, kappa_optimizer.x)\n    tuned_kappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"Best QWK SCORE : {tuned_kappa:.3f}\")\n\n    # Generate test predictions\n    test_pred_mean = test_predictions.mean(axis=1)\n    test_pred_tuned = threshold_rounder(test_pred_mean, kappa_optimizer.x)\n\n    submission = pd.DataFrame({\n        'id': sample_sub['id'],\n        'sii': test_pred_tuned\n    })\n\n    return submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.622049Z","iopub.status.idle":"2024-12-22T05:57:36.622316Z","shell.execute_reply.started":"2024-12-22T05:57:36.622183Z","shell.execute_reply":"2024-12-22T05:57:36.622197Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Light_params = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'n_estimators': 300,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  \n    'lambda_l2': 0.01,\n    'random_state': SEED,\n    'device': 'gpu'\n}\nLight_Model = LGBMRegressor(**Light_params, verbose=-1)\n# Light_submission = train_model(Light_Model, train_data, test_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.623079Z","iopub.status.idle":"2024-12-22T05:57:36.623343Z","shell.execute_reply.started":"2024-12-22T05:57:36.623212Z","shell.execute_reply":"2024-12-22T05:57:36.623226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"XGB_params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1, \n    'reg_lambda': 5, \n    'random_state': SEED,\n    'tree_method': 'gpu_hist',\n}\nXGB_Model = XGBRegressor(**XGB_params)\n# XGB_submission = train_model(XGB_Model, train_data, test_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.624539Z","iopub.status.idle":"2024-12-22T05:57:36.624807Z","shell.execute_reply.started":"2024-12-22T05:57:36.624677Z","shell.execute_reply":"2024-12-22T05:57:36.624691Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"CatBoost_params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    #'cat_features': category_cols,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10,  \n    'task_type': 'GPU'\n}\nCat_Model = CatBoostRegressor(**CatBoost_params)\n# Cat_submission = train_model(Cat_Model, train_data, test_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.625976Z","iopub.status.idle":"2024-12-22T05:57:36.626273Z","shell.execute_reply.started":"2024-12-22T05:57:36.626130Z","shell.execute_reply":"2024-12-22T05:57:36.626147Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class TabNetCheckpoint(Callback):\n    def __init__(self, filepath, monitor='val_loss', mode='min', save_best_only=True, verbose=1):\n        super().__init__()\n        self.filepath = filepath\n        self.monitor = monitor\n        self.mode = mode\n        self.save_best_only = save_best_only\n        self.verbose = verbose\n        self.best = float('inf') if mode == 'min' else -float('inf')\n        \n    def on_train_begin(self, logs=None):\n        self.model = self.trainer\n        \n    def on_epoch_end(self, epoch, logs=None):\n        logs = logs or {}\n        current = logs.get(self.monitor)\n        if current is None:\n            return\n        \n        if (self.mode == 'min' and current < self.best) or (self.mode == 'max' and current > self.best):\n            if self.verbose:\n                print(f'\\nEpoch {epoch}: {self.monitor} improved from {self.best:.4f} to {current:.4f}')\n            self.best = current\n            if self.save_best_only:\n                self.model.save_model(self.filepath)\n\nclass TabNet(BaseEstimator, RegressorMixin):\n    def __init__(self, **kwargs):\n        self.model = TabNetRegressor(**kwargs)\n        self.kwargs = kwargs\n        self.imputer = SimpleImputer(strategy='median')\n        self.best_model_path = 'best_tabnet.pt'\n        \n    def fit(self, X, y):\n        X_imputed = self.imputer.fit_transform(X)\n            \n        if hasattr(y, 'values'):\n            y = y.values\n\n        X_train, X_valid, y_train, y_valid = train_test_split(X_imputed, y, test_size=0.2, random_state=SEED)\n\n        history = self.model.fit(\n            X_train = X_train,\n            y_train = y_train.reshape(-1, 1),\n            eval_set = [(X_valid, y_valid.reshape(-1, 1))],\n            eval_name = ['valid'],\n            eval_metric = ['mse'],\n            max_epochs = 200,\n            patience = 20,\n            batch_size = 1024,\n            virtual_batch_size = 128,\n            num_workers = 0,\n            drop_last = False,\n            callbacks = [\n                TabNetCheckpoint(\n                    filepath = self.best_model_path,\n                    monitor = 'valid_mse',\n                    mode = 'min',\n                    save_best_only = True,\n                    verbose = True\n                )\n            ]\n        )\n\n        if os.path.exists(self.best_model_path):\n            self.model.load_model(self.best_model_path)\n            os.remove(self.best_model_path)\n\n        return self\n\n    def predict(self, X):\n        X_imputed = self.imputer.transform(X)\n        return self.model.predict(X_imputed).flatten()\n\n    def __deepcopy__(self, memo):\n        cls = self.__class__\n        result = self.__new__(cls)\n        memo[id(self)] = result\n        for k, v in self.__dict__.items():\n            setattr(result, k, deepcopy(v, memo))\n\n        return result\n\nTabNet_Params = {\n    'n_d': 64,\n    'n_a': 64, \n    'n_steps': 5,           \n    'gamma': 1.5,          \n    'n_independent': 2,     \n    'n_shared': 2,          \n    'lambda_sparse': 1e-4,  \n    'optimizer_fn': torch.optim.AdamW,\n    'optimizer_params': dict(lr=0.02, weight_decay=1e-5),\n    'mask_type': 'entmax',\n    'scheduler_params': dict(mode=\"min\", patience=10, min_lr=1e-5, factor=0.2),\n    'scheduler_fn': torch.optim.lr_scheduler.ReduceLROnPlateau,\n    'device_name': 'cuda' if torch.cuda.is_available() else 'cpu'\n}\nTabNet_Model = TabNet(**TabNet_Params)\n# tabnet_sub = train_model(TabNet_Model, train_data, test_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.627374Z","iopub.status.idle":"2024-12-22T05:57:36.627678Z","shell.execute_reply.started":"2024-12-22T05:57:36.627543Z","shell.execute_reply":"2024-12-22T05:57:36.627558Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model1 = VotingRegressor(estimators=[\n    ('lightgbm', Light_Model),\n    ('xgboost', XGB_Model),\n    ('catboost', Cat_Model),])\n    # ('tabnet', TabNet_Model)\n# ], weights=[4.0, 4.0, 5.0, 4.0])\n\nsubmission1 = train_model(model1, train_data, test_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.629273Z","iopub.status.idle":"2024-12-22T05:57:36.629706Z","shell.execute_reply.started":"2024-12-22T05:57:36.629484Z","shell.execute_reply":"2024-12-22T05:57:36.629508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission1.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.630850Z","iopub.status.idle":"2024-12-22T05:57:36.631277Z","shell.execute_reply.started":"2024-12-22T05:57:36.631046Z","shell.execute_reply":"2024-12-22T05:57:36.631069Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Solution 2: Không encode time series data","metadata":{}},{"cell_type":"code","source":"train = pd.merge(train_csv, ts_train, how=\"left\", on='id')\ntest = pd.merge(test_csv, ts_test, how=\"left\", on='id')\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += list(df_train_ts.columns)\ntrain = train[featuresCols]\n\ntrain = update(train)\ntest = update(test)\n\nfor col in category_cols:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ntrain = feature_engineering(train)\ntest = feature_engineering(test)\ntrain = train.dropna(thresh=10, axis=0)\ntrain_data = train.dropna(subset='sii', axis=0)\n\nif np.any(np.isinf(train_data)):\n    train_data = train_data.replace([np.inf, -np.inf], np.nan)\n\ntestCols = list(train_data.columns)\ntestCols.remove('sii')\ntest_data = test[testCols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.632278Z","iopub.status.idle":"2024-12-22T05:57:36.632583Z","shell.execute_reply.started":"2024-12-22T05:57:36.632431Z","shell.execute_reply":"2024-12-22T05:57:36.632455Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"CatBoost_params2 = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10,  \n    'task_type': 'GPU'\n}\nCat_Model2 = CatBoostRegressor(**CatBoost_params2)\n\nmodel2 = VotingRegressor(estimators=[\n    ('lgb', Pipeline(steps=[('imputer', sim_imputer), ('regressor', Light_Model)])),\n    ('xgb', Pipeline(steps=[('imputer', sim_imputer), ('regressor', XGB_Model)])),\n    ('cat', Pipeline(steps=[('imputer', sim_imputer), ('regressor', Cat_Model2)])),\n    ('rf', Pipeline(steps=[('imputer', sim_imputer), ('regressor', RandomForestRegressor(random_state=SEED))])),\n    ('gb', Pipeline(steps=[('imputer', sim_imputer), ('regressor', GradientBoostingRegressor(random_state=SEED))]))\n])\nsubmission2 = train_model(model2, train_data, test_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.634245Z","iopub.status.idle":"2024-12-22T05:57:36.634696Z","shell.execute_reply.started":"2024-12-22T05:57:36.634466Z","shell.execute_reply":"2024-12-22T05:57:36.634492Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Solution 3: Xử lý time series data đơn giản","metadata":{}},{"cell_type":"code","source":"def process_file(filename, dirname):\n    \"\"\"Xử lý file time series.\"\"\"\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname):\n    \"\"\"Tạo time series df với multithreading.\"\"\"\n    ids = os.listdir(dirname)\n\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n\n    stats, indexes = zip(*results)\n\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n\n    return df\n\nts_train = load_time_series('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet')\nts_test = load_time_series('/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet')\ndf_train_ts = ts_train.drop('id', axis=1)\ndf_test_ts = ts_test.drop('id', axis=1)\nts_train_encoded = perform_autoencoder(df_train_ts)\nts_test_encoded = perform_autoencoder(df_test_ts)\nenc_time_series_cols = ts_train_encoded.columns.tolist()\nts_train_encoded[\"id\"]=ts_train[\"id\"]\nts_test_encoded[\"id\"]=ts_test[\"id\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.635644Z","iopub.status.idle":"2024-12-22T05:57:36.636063Z","shell.execute_reply.started":"2024-12-22T05:57:36.635840Z","shell.execute_reply":"2024-12-22T05:57:36.635865Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.merge(train_csv, ts_train_encoded, how=\"left\", on='id')\ntest = pd.merge(test_csv, ts_test_encoded, how=\"left\", on='id')\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += enc_time_series_cols\n\ntrain = train[featuresCols]\n\n# numeric_cols = list(train.select_dtypes(include=['float64', 'int64']).columns)\n# # imputed_data = sim_imputer.fit_transform(train[numeric_cols])\n# imputed_data = knn_imputer.fit_transform(train[numeric_cols])\n# train_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\n# train_imputed['sii'] = train_imputed['sii'].round().astype(int)\n# for col in train.columns:\n#     if col not in numeric_cols:\n#         train_imputed[col] = train[col]       \n# train = train_imputed\n\ntrain = update(train)\ntest = update(test)\n\nfor col in category_cols:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ntrain = feature_engineering(train)\ntest = feature_engineering(test)\ntrain = train.dropna(thresh=10, axis=0)\ntrain_data = train.dropna(subset='sii', axis=0)\n\nif np.any(np.isinf(train_data)):\n    train_data = train_data.replace([np.inf, -np.inf], np.nan)\n\ntestCols = list(train_data.columns)\ntestCols.remove('sii')\ntest_data = test[testCols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.637743Z","iopub.status.idle":"2024-12-22T05:57:36.638044Z","shell.execute_reply.started":"2024-12-22T05:57:36.637903Z","shell.execute_reply":"2024-12-22T05:57:36.637919Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model3 = VotingRegressor(estimators=[\n    ('lightgbm', Light_Model),\n    ('xgboost', XGB_Model),\n    ('catboost', Cat_Model),\n])\n\nsubmission3 = train_model(model3, train_data, test_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.638901Z","iopub.status.idle":"2024-12-22T05:57:36.639165Z","shell.execute_reply.started":"2024-12-22T05:57:36.639033Z","shell.execute_reply":"2024-12-22T05:57:36.639047Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"submission1 = submission1.sort_values(by='id').reset_index(drop=True)\nsubmission2 = submission2.sort_values(by='id').reset_index(drop=True)\nsubmission3 = submission3.sort_values(by='id').reset_index(drop=True)\n\ncombined = pd.DataFrame({\n    'id': submission1['id'],\n    'sii_1': submission1['sii'],\n    'sii_2': submission2['sii'],\n    'sii_3': submission3['sii']\n})\n\ndef majority_vote(row):\n    return row.mode()[0]\n\ncombined['final_sii'] = combined[['sii_1', 'sii_2', 'sii_3']].apply(majority_vote, axis=1)\nfinal_submission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\nfinal_submission.to_csv('submission.csv', index=False)\n\nfinal_submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T05:57:36.640274Z","iopub.status.idle":"2024-12-22T05:57:36.640588Z","shell.execute_reply.started":"2024-12-22T05:57:36.640443Z","shell.execute_reply":"2024-12-22T05:57:36.640460Z"}},"outputs":[],"execution_count":null}]}