{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import Library\n","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport polars as pl\n\nfrom sklearn.impute import KNNImputer\nfrom sklearn.impute import SimpleImputer","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:24:04.078731Z","iopub.execute_input":"2024-12-17T20:24:04.079251Z","iopub.status.idle":"2024-12-17T20:24:04.086260Z","shell.execute_reply.started":"2024-12-17T20:24:04.079216Z","shell.execute_reply":"2024-12-17T20:24:04.084878Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Database(csv) analysis and processing","metadata":{}},{"cell_type":"code","source":"data_train = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\ndata_test = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")\ndata_dictionary = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:24:04.088948Z","iopub.execute_input":"2024-12-17T20:24:04.089650Z","iopub.status.idle":"2024-12-17T20:24:04.144293Z","shell.execute_reply.started":"2024-12-17T20:24:04.089597Z","shell.execute_reply":"2024-12-17T20:24:04.143141Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:24:04.145642Z","iopub.execute_input":"2024-12-17T20:24:04.145964Z","iopub.status.idle":"2024-12-17T20:24:04.153859Z","shell.execute_reply.started":"2024-12-17T20:24:04.145934Z","shell.execute_reply":"2024-12-17T20:24:04.152553Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:24:04.156158Z","iopub.execute_input":"2024-12-17T20:24:04.156536Z","iopub.status.idle":"2024-12-17T20:24:04.170466Z","shell.execute_reply.started":"2024-12-17T20:24:04.156475Z","shell.execute_reply":"2024-12-17T20:24:04.169045Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:24:04.172134Z","iopub.execute_input":"2024-12-17T20:24:04.172574Z","iopub.status.idle":"2024-12-17T20:24:04.204064Z","shell.execute_reply.started":"2024-12-17T20:24:04.172535Z","shell.execute_reply":"2024-12-17T20:24:04.202629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:24:04.205877Z","iopub.execute_input":"2024-12-17T20:24:04.206319Z","iopub.status.idle":"2024-12-17T20:24:04.238579Z","shell.execute_reply.started":"2024-12-17T20:24:04.206283Z","shell.execute_reply":"2024-12-17T20:24:04.237247Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Show target column**","metadata":{}},{"cell_type":"code","source":"filtered_data = data_train[data_train['sii'].notna()]\nprint(filtered_data['sii'].count())\nprint(filtered_data)\n\nsns.countplot(x='sii', data=filtered_data)\nplt.title('Phân phối cột mục tiêu (sii)')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:24:04.239889Z","iopub.execute_input":"2024-12-17T20:24:04.240201Z","iopub.status.idle":"2024-12-17T20:24:04.437854Z","shell.execute_reply.started":"2024-12-17T20:24:04.240171Z","shell.execute_reply":"2024-12-17T20:24:04.436764Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('Columns missing in test:')\nprint([f for f in data_train.columns if f not in data_test.columns])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:24:04.439201Z","iopub.execute_input":"2024-12-17T20:24:04.439583Z","iopub.status.idle":"2024-12-17T20:24:04.446795Z","shell.execute_reply.started":"2024-12-17T20:24:04.439548Z","shell.execute_reply":"2024-12-17T20:24:04.445169Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Show and handle missing values¶\r- \nDrop features with the rate of missing values >= 30%","metadata":{}},{"cell_type":"code","source":"missing_values = filtered_data.isnull().sum()\nprint(\"Số lượng giá trị thiếu:\\n\", missing_values[missing_values >= 0])\nmissing_percent = (missing_values / len(data_train)) * 100\n\nmissing_data = pd.DataFrame({\n    'Số lượng thiếu': missing_values,\n    'Phần trăm (%)': missing_percent\n})\n\nmissing_data = missing_data[missing_data['Số lượng thiếu'] >= 0]\n\nmissing_data_sort = missing_data.sort_values(by='Phần trăm (%)', ascending=True)\n\nif missing_data_sort.empty:\n    print(\"Không có cột nào có giá trị thiếu.\")\nelse:\n    plt.figure(figsize=(10, len(missing_data) * 0.2))\n    sns.barplot(\n        y=missing_data_sort.index, \n        x=missing_data_sort['Phần trăm (%)'], \n        palette=\"viridis\"\n    )\n    plt.xticks(rotation=45, ha='right')\n    plt.title(\"Tỷ lệ phần trăm giá trị thiếu theo cột\", fontsize=14)\n    plt.ylabel(\"Feature\", fontsize=12)\n    plt.xlabel(\"Phần trăm (%)\", fontsize=12)\n\n    plt.gca().yaxis.set_tick_params(labelsize=10, pad=5)\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:24:04.450793Z","iopub.execute_input":"2024-12-17T20:24:04.451255Z","iopub.status.idle":"2024-12-17T20:24:05.572597Z","shell.execute_reply.started":"2024-12-17T20:24:04.451217Z","shell.execute_reply":"2024-12-17T20:24:05.571379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"columns_to_drop = missing_data[missing_data['Phần trăm (%)'] > 30].index\n\nprint(\"Các feature có tỷ lệ giá trị thiếu lớn hơn 30%:\")\nprint(columns_to_drop)\n\ndata_train_cleaned = data_train.drop(columns=columns_to_drop)\n\ndata_train_cleaned.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:24:05.574311Z","iopub.execute_input":"2024-12-17T20:24:05.574798Z","iopub.status.idle":"2024-12-17T20:24:05.600269Z","shell.execute_reply.started":"2024-12-17T20:24:05.574750Z","shell.execute_reply":"2024-12-17T20:24:05.599045Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- **Shows the relationship between a few features**\n- The chart of Age Distribution\n- The chart of Sex Distribution","metadata":{}},{"cell_type":"code","source":"print(data_train[['Basic_Demos-Age', 'Basic_Demos-Sex']].describe())\nsns.histplot(data_train['Basic_Demos-Age'], kde=True)\nplt.title('Age Distribution')\nplt.show()\nsns.countplot(x='Basic_Demos-Sex', data=data_train)\nplt.title('Sex Distribution')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:24:05.601523Z","iopub.execute_input":"2024-12-17T20:24:05.601973Z","iopub.status.idle":"2024-12-17T20:24:06.112498Z","shell.execute_reply.started":"2024-12-17T20:24:05.601926Z","shell.execute_reply":"2024-12-17T20:24:06.111325Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Fill databases and change types for a few features","metadata":{}},{"cell_type":"code","source":"# Use KNNImputer to fill features with types as float, int\nknn_imputer = KNNImputer(n_neighbors=5)\n\nnum_cols = data_train_cleaned.select_dtypes(include=['float64', 'int64']).columns\n\ndata_train_cleaned[num_cols] = knn_imputer.fit_transform(data_train_cleaned[num_cols])\n\nprint(data_train_cleaned.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:24:06.113731Z","iopub.execute_input":"2024-12-17T20:24:06.114145Z","iopub.status.idle":"2024-12-17T20:24:12.246453Z","shell.execute_reply.started":"2024-12-17T20:24:06.114112Z","shell.execute_reply":"2024-12-17T20:24:12.245297Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Use SimpleImputer to fill features with types as object\ncat_cols = data_train_cleaned.select_dtypes(include=['object']).columns\ncat_imputer = SimpleImputer(strategy='most_frequent')\n\ndata_train_cleaned[cat_cols] = cat_imputer.fit_transform(data_train_cleaned[cat_cols])\n\nprint(data_train_cleaned.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:24:12.248017Z","iopub.execute_input":"2024-12-17T20:24:12.248368Z","iopub.status.idle":"2024-12-17T20:24:12.274508Z","shell.execute_reply.started":"2024-12-17T20:24:12.248331Z","shell.execute_reply":"2024-12-17T20:24:12.273451Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_values_train_after = data_train_cleaned.isnull().sum()\nprint(\"Missing values after imputation in train dataset:\\n\", missing_values_train_after)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:24:12.275754Z","iopub.execute_input":"2024-12-17T20:24:12.276088Z","iopub.status.idle":"2024-12-17T20:24:12.291540Z","shell.execute_reply.started":"2024-12-17T20:24:12.276056Z","shell.execute_reply":"2024-12-17T20:24:12.290373Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Show the relatives between features and target column**","metadata":{}},{"cell_type":"code","source":"num_cols = ['Basic_Demos-Age', 'Physical-BMI', 'PCIAT-PCIAT_Total']\nfor col in num_cols:\n    sns.boxplot(x='sii', y=col, data=data_train_cleaned)\n    plt.title(f'Mối quan hệ giữa {col} và sii')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:24:12.293040Z","iopub.execute_input":"2024-12-17T20:24:12.293434Z","iopub.status.idle":"2024-12-17T20:24:13.715313Z","shell.execute_reply.started":"2024-12-17T20:24:12.293401Z","shell.execute_reply":"2024-12-17T20:24:13.714165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_cols = ['Basic_Demos-Sex', 'Basic_Demos-Enroll_Season', 'PCIAT-Season']\nfor col in cat_cols:\n    sns.countplot(x=col, hue='sii', data=data_train_cleaned)\n    plt.title(f'Mối quan hệ giữa {col} và sii')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:24:13.716766Z","iopub.execute_input":"2024-12-17T20:24:13.717127Z","iopub.status.idle":"2024-12-17T20:24:14.888856Z","shell.execute_reply.started":"2024-12-17T20:24:13.717089Z","shell.execute_reply":"2024-12-17T20:24:14.887741Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Convert object data to float/int data**","metadata":{}},{"cell_type":"code","source":"season_mapping = {'Spring': 1, 'Summer': 2, 'Fall': 3, 'Winter': 4}\nseason_cols = [col for col in data_train_cleaned.columns if 'Season' in col]\nfor col in season_cols:\n    data_train_cleaned[col] = data_train_cleaned[col].map(season_mapping)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:24:14.890470Z","iopub.execute_input":"2024-12-17T20:24:14.891545Z","iopub.status.idle":"2024-12-17T20:24:14.906553Z","shell.execute_reply.started":"2024-12-17T20:24:14.891472Z","shell.execute_reply":"2024-12-17T20:24:14.904963Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop column id\ndata_train_no_id = data_train_cleaned.drop(columns = ['id'], errors = 'ignore')\ndata_train_no_id.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:24:14.908140Z","iopub.execute_input":"2024-12-17T20:24:14.908614Z","iopub.status.idle":"2024-12-17T20:24:14.940265Z","shell.execute_reply.started":"2024-12-17T20:24:14.908566Z","shell.execute_reply":"2024-12-17T20:24:14.939018Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop columns PCIAT-PCIAT\npciat_columns = [col for col in data_train_cleaned.columns if 'PCIAT-PCIAT' in col and col != 'PCIAT-PCIAT_Total']\ncorr_with_total = data_train_cleaned[pciat_columns].corrwith(data_train_cleaned['PCIAT-PCIAT_Total'])\nprint(\"Mối tương quan với PCIAT_PCIAT_TOTAL:\")\nprint(corr_with_total)\n\ndata_train_no_id.drop(columns= pciat_columns, inplace=True)\ndata_train_cleaned.drop(columns= pciat_columns, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:24:14.941744Z","iopub.execute_input":"2024-12-17T20:24:14.942158Z","iopub.status.idle":"2024-12-17T20:24:14.966722Z","shell.execute_reply.started":"2024-12-17T20:24:14.942125Z","shell.execute_reply":"2024-12-17T20:24:14.965452Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineering(df):\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    df['BMI_PHR'] = df['Physical-BMI'] * df['Physical-HeartRate']\n    \n    return df\n\ndata_train_no_id = feature_engineering(data_train_no_id)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:24:14.968608Z","iopub.execute_input":"2024-12-17T20:24:14.969075Z","iopub.status.idle":"2024-12-17T20:24:14.989377Z","shell.execute_reply.started":"2024-12-17T20:24:14.969025Z","shell.execute_reply":"2024-12-17T20:24:14.988221Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- **Show heatmap of database**\n- **Drop feature with threshold > 0.8**","metadata":{}},{"cell_type":"code","source":"corr_matrix = data_train_no_id.corr()\n\nplt.figure(figsize=(30, 30))\nsns.heatmap(corr_matrix, annot=True, cmap='coolwarm', linewidths=0.5, fmt='.2f', vmin=-1, vmax=1)\nplt.title('Heatmap of Correlation Matrix')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:24:14.990810Z","iopub.execute_input":"2024-12-17T20:24:14.991162Z","iopub.status.idle":"2024-12-17T20:24:23.519081Z","shell.execute_reply.started":"2024-12-17T20:24:14.991128Z","shell.execute_reply":"2024-12-17T20:24:23.517656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"threshold = 0.8\n\nto_drop = set()\nfor i in range(len(corr_matrix.columns)):\n    for j in range(i):\n        if abs(corr_matrix.iloc[i, j]) > threshold:\n            colname = corr_matrix.columns[i]\n            to_drop.add(colname)\n\nto_drop.discard('sii')\n\ndata_train_cleaned = feature_engineering(data_train_cleaned)\ndata_train_cleaned = data_train_cleaned.drop(columns=to_drop)\n\nprint(f\"Những cột đã bị loại bỏ: {to_drop}\")\nprint(data_train_cleaned.shape)  \n\ndata_train_cleaned = data_train_cleaned.drop(columns=['PCIAT-Season', 'PCIAT-PCIAT_Total'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:24:23.520639Z","iopub.execute_input":"2024-12-17T20:24:23.520984Z","iopub.status.idle":"2024-12-17T20:24:23.590427Z","shell.execute_reply.started":"2024-12-17T20:24:23.520952Z","shell.execute_reply":"2024-12-17T20:24:23.589349Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Database(parquet) analysis and processing - Actigraphy (time series)\n","metadata":{}},{"cell_type":"code","source":"actigraphy = pl.read_parquet('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id=0417c91e/part-0.parquet')\nactigraphy","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:24:23.591558Z","iopub.execute_input":"2024-12-17T20:24:23.591863Z","iopub.status.idle":"2024-12-17T20:24:23.620187Z","shell.execute_reply.started":"2024-12-17T20:24:23.591830Z","shell.execute_reply":"2024-12-17T20:24:23.619065Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\nimport os\nimport torch\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nfrom sklearn.preprocessing import StandardScaler\nimport torch.nn as nn\nimport torch.optim as optim\n\n# Seed for reproducibility\ndef seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True\nseed_everything(2024)\n\n# Function to load and process files\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\n# Optimized loading time series\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n\n# AutoEncoder class\nclass AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim * 3), nn.ReLU(),\n            nn.Linear(encoding_dim * 3, encoding_dim * 2), nn.ReLU(),\n            nn.Linear(encoding_dim * 2, encoding_dim), nn.ReLU()\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim * 2), nn.ReLU(),\n            nn.Linear(input_dim * 2, input_dim * 3), nn.ReLU(),\n            nn.Linear(input_dim * 3, input_dim), nn.Sigmoid()\n        )\n\n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded\n\n# Optimized Autoencoder Training Function\ndef perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    data_tensor = torch.FloatTensor(df_scaled)\n    \n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(autoencoder.parameters())\n    \n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i:i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n            \n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}')\n    \n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n    return df_encoded\n\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:24:23.625673Z","iopub.execute_input":"2024-12-17T20:24:23.626164Z","iopub.status.idle":"2024-12-17T20:25:46.677230Z","shell.execute_reply.started":"2024-12-17T20:24:23.626130Z","shell.execute_reply":"2024-12-17T20:25:46.675956Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\n\ntrain_ts_encoded = perform_autoencoder(df_train, encoding_dim=60, epochs=100, batch_size=32)\ntest_ts_encoded = perform_autoencoder(df_test, encoding_dim=60, epochs=100, batch_size=32)\n\ntrain_ts_encoded[\"id\"] = train_ts[\"id\"]\ntest_ts_encoded['id'] = test_ts[\"id\"]\n\ndata_cleaned = pd.merge(data_train_cleaned, train_ts_encoded, how=\"left\", on='id')\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:25:46.678978Z","iopub.execute_input":"2024-12-17T20:25:46.679465Z","iopub.status.idle":"2024-12-17T20:25:57.616754Z","shell.execute_reply.started":"2024-12-17T20:25:46.679416Z","shell.execute_reply":"2024-12-17T20:25:57.615577Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_cleaned","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:25:57.618175Z","iopub.execute_input":"2024-12-17T20:25:57.618633Z","iopub.status.idle":"2024-12-17T20:25:57.650069Z","shell.execute_reply.started":"2024-12-17T20:25:57.618583Z","shell.execute_reply":"2024-12-17T20:25:57.649073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create new data for test\ntest = feature_engineering(test)\n\ntest","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:25:57.651162Z","iopub.execute_input":"2024-12-17T20:25:57.651459Z","iopub.status.idle":"2024-12-17T20:25:57.697023Z","shell.execute_reply.started":"2024-12-17T20:25:57.651429Z","shell.execute_reply":"2024-12-17T20:25:57.695996Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"knn_imputer = KNNImputer(n_neighbors=5)\n\nnum_cols = data_cleaned.select_dtypes(include=['int32', 'int64', 'float64', 'int64']).columns\n\ndata_cleaned[num_cols] = knn_imputer.fit_transform(data_cleaned[num_cols])\n\ndata_cleaned = data_cleaned.dropna(thresh=10, axis=0)\n\nprint(data_cleaned.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:25:57.698227Z","iopub.execute_input":"2024-12-17T20:25:57.698557Z","iopub.status.idle":"2024-12-17T20:25:57.724927Z","shell.execute_reply.started":"2024-12-17T20:25:57.698525Z","shell.execute_reply":"2024-12-17T20:25:57.723842Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_cleaned.drop('id', axis=1)\ndata_cleaned","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:25:57.726335Z","iopub.execute_input":"2024-12-17T20:25:57.726758Z","iopub.status.idle":"2024-12-17T20:25:57.761012Z","shell.execute_reply.started":"2024-12-17T20:25:57.726724Z","shell.execute_reply":"2024-12-17T20:25:57.759949Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_cleaned.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:25:57.762307Z","iopub.execute_input":"2024-12-17T20:25:57.762644Z","iopub.status.idle":"2024-12-17T20:25:57.780368Z","shell.execute_reply.started":"2024-12-17T20:25:57.762612Z","shell.execute_reply":"2024-12-17T20:25:57.778976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef filter_test_data(train_data, test_data, target_column):\n    train_columns = [col for col in train_data.columns if col != target_column]\n    \n    test_data_filtered = test_data[train_columns]\n    \n    return test_data_filtered\n\ntest = filter_test_data (data_cleaned, test, 'sii')\nseason_mapping = {'Spring': 1, 'Summer': 2, 'Fall': 3, 'Winter': 4}\nseason_cols = [col for col in test.columns if 'Season' in col]\nfor col in season_cols:\n    test[col] = test[col].map(season_mapping)\ntest.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:25:57.781814Z","iopub.execute_input":"2024-12-17T20:25:57.782182Z","iopub.status.idle":"2024-12-17T20:25:57.806682Z","shell.execute_reply.started":"2024-12-17T20:25:57.782149Z","shell.execute_reply":"2024-12-17T20:25:57.805547Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training model ","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import cohen_kappa_score  \nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.metrics import make_scorer\nfrom scipy.optimize import minimize\nfrom sklearn.model_selection import KFold\nfrom tqdm import tqdm\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    rounded_y_true = threshold_Rounder(y_true, [0.5, 1.5, 2.5])\n    rounded_y_pred = threshold_Rounder(y_pred, [0.5, 1.5, 2.5]) \n\n    return cohen_kappa_score(rounded_y_true, rounded_y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_y_true = threshold_Rounder(y_true, thresholds)\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(rounded_y_true, rounded_p)\n\n\ndef TrainML(train_data, test_data, target_column, n_splits=5, random_state=42):\n    X = train_data.drop(columns=[target_column, 'id'])\n    y = train_data[target_column]\n    test_data_dropped = test_data.drop(columns=['id'])\n\n    test_ids = test_data['id']\n\n    from sklearn.impute import SimpleImputer  \n    imputer = SimpleImputer(strategy='mean')\n    X = imputer.fit_transform(X)\n    test_data_dropped = imputer.transform(test_data_dropped)\n\n    # Khởi tạo Cross Validation\n    SKF = KFold(n_splits=n_splits, shuffle=True, random_state=random_state)\n\n    # Lưu kết quả\n    oof_non_rounded = np.zeros(len(y), dtype=float)\n    test_preds = np.zeros((len(test_data), n_splits))\n\n    # Model\n    models = {\n        'xgboost': XGBRegressor(\n            learning_rate=0.05, max_depth=6, n_estimators=200, subsample=0.8,\n            colsample_bytree=0.8, reg_alpha=1, reg_lambda=5, random_state=random_state\n        ),\n        'gboost': GradientBoostingRegressor(\n            learning_rate=0.05, max_depth=6, n_estimators=200, random_state=random_state\n        ),\n        'catboost': CatBoostRegressor(\n            learning_rate=0.05, depth=6, iterations=200, random_state=random_state, verbose=0  # Cấu hình CatBoost\n        )\n    }\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X[train_idx], X[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        # Huấn luyện từng model\n        for model_name, model in models.items():\n            model.fit(X_train, y_train)\n            y_val_pred = model.predict(X_val)\n            oof_non_rounded[test_idx] += y_val_pred / len(models)\n\n            # Dự đoán trên tập test\n            test_preds[:, fold] += model.predict(test_data_dropped) / len(models)\n\n    # Tối ưu threshold\n    KappaOptimizer = minimize(\n        evaluate_predictions, x0=[0.5, 1.5, 2.5],\n        args=(y, oof_non_rounded), method='Nelder-Mead'\n    )\n    assert KappaOptimizer.success, \"Optimization did not converge.\"\n\n    # Áp dụng threshold tối ưu\n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOptimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {tKappa:.3f}\")\n\n    # Dự đoán tập test\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOptimizer.x)\n\n    # Tạo submission với id\n    submission = pd.DataFrame({\n        'id': test_ids,  # Sử dụng lại cột 'id' đã lưu\n        target_column: tpTuned\n    })\n\n    return submission\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:25:57.807998Z","iopub.execute_input":"2024-12-17T20:25:57.808344Z","iopub.status.idle":"2024-12-17T20:25:57.827684Z","shell.execute_reply.started":"2024-12-17T20:25:57.808313Z","shell.execute_reply":"2024-12-17T20:25:57.826605Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"submission = TrainML(data_cleaned, test, target_column=\"sii\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:25:57.828954Z","iopub.execute_input":"2024-12-17T20:25:57.829290Z","iopub.status.idle":"2024-12-17T20:27:11.870021Z","shell.execute_reply.started":"2024-12-17T20:25:57.829259Z","shell.execute_reply":"2024-12-17T20:27:11.868808Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:27:11.872292Z","iopub.execute_input":"2024-12-17T20:27:11.872786Z","iopub.status.idle":"2024-12-17T20:27:11.883042Z","shell.execute_reply.started":"2024-12-17T20:27:11.872736Z","shell.execute_reply":"2024-12-17T20:27:11.882001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T20:27:11.884226Z","iopub.execute_input":"2024-12-17T20:27:11.884544Z","iopub.status.idle":"2024-12-17T20:27:11.896560Z","shell.execute_reply.started":"2024-12-17T20:27:11.884515Z","shell.execute_reply":"2024-12-17T20:27:11.895453Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}