{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":7453542,"sourceType":"datasetVersion","datasetId":921302}],"dockerImageVersionId":30761,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div style=\"display: flex; align-items: center; justify-content: flex-start; text-align: left; width: fit-content; margin: 0 auto;\">\n    <img src=\"https://media.giphy.com/media/LO8oXHPum0xworIyk4/giphy.gif?cid=ecf05e47mbgvh6ylsgvcjv4motlmhj5eqzukcs5tg9kltdn3&ep=v1_gifs_search&rid=giphy.gif&ct=g\"\n         style=\"max-width: 30px; margin-right: 10px;\">\n    <span>Starting from 7th notebook versions I use recalculated SII scores (see explanations below).</span>\n</div>","metadata":{}},{"cell_type":"markdown","source":"\n* train, test,  data_dict: nguyên mẫu dữ liệu\n* train_cleaned: đã lọc các hàng có sii = NaN, Xu  \n* train_mean, train_median, train_zero: Thay thế NaN bằng mean, median, 0\n\n\n ","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport polars as pl\nimport polars.selectors as cs\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns\n\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:17:11.65685Z","iopub.execute_input":"2024-11-30T02:17:11.65719Z","iopub.status.idle":"2024-11-30T02:17:11.664608Z","shell.execute_reply.started":"2024-11-30T02:17:11.657161Z","shell.execute_reply":"2024-11-30T02:17:11.663602Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"warnings.filterwarnings('ignore', category=FutureWarning)\n\nsns.set(style=\"whitegrid\")\n%matplotlib inline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:17:11.666129Z","iopub.execute_input":"2024-11-30T02:17:11.666402Z","iopub.status.idle":"2024-11-30T02:17:11.682383Z","shell.execute_reply.started":"2024-11-30T02:17:11.666375Z","shell.execute_reply":"2024-11-30T02:17:11.681628Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Understanding the task","metadata":{}},{"cell_type":"markdown","source":"Mục tiêu của cuộc thi này là dự đoán Severity Impairment Index (sii), chỉ số đánh giá mức độ sử dụng internet có vấn đề ở trẻ em và thanh thiếu niên, dựa trên dữ liệu hoạt động thể chất và các đặc trưng khác.\n\nsii được tính từ PCIAT-PCIAT_Total, tổng điểm của Parent-Child Internet Addiction Test (PCIAT: 20 câu hỏi, được chấm từ 0-5 điểm).","metadata":{}},{"cell_type":"markdown","source":"Biến mục tiêu (sii) được định nghĩa như sau:\n\n0: Không có (PCIAT-PCIAT_Total từ 0 đến 30)\n1: Nhẹ (PCIAT-PCIAT_Total từ 31 đến 49)\n2: Vừa (PCIAT-PCIAT_Total từ 50 đến 79)\n3: Nghiêm trọng (PCIAT-PCIAT_Total từ 80 trở lên)\nĐiều này làm cho sii trở thành một biến phân loại thứ tự với bốn cấp độ, trong đó thứ tự của các danh mục có ý nghĩa.","metadata":{}},{"cell_type":"markdown","source":"Các loại bài toán Machine Learning có thể áp dụng với sii làm biến mục tiêu:\n\n1. Phân loại thứ tự (Ordinal classification)\nSử dụng các mô hình như hồi quy logistic thứ tự hoặc các mô hình với hàm mất mát tùy chỉnh cho phân loại thứ tự.\n\n2. Phân loại đa lớp (Multiclass classification)\nXem sii như một biến phân loại không có thứ tự (nominal), không xét đến mối quan hệ thứ tự giữa các mức.\n\n3. Hồi quy (Regression)\nXử lý sii như một biến liên tục, bỏ qua tính chất rời rạc của các danh mục. Sau đó làm tròn dự đoán để xác định cấp độ sii.\n\n4. Tùy chỉnh (Custom)\nThiết kế các hàm mất mát đặc biệt, phạt nặng hơn khi dự đoán sai lệch giữa các cấp độ cách xa nhau (ví dụ: dự đoán sai từ 0 lên 3 bị phạt nặng hơn từ 0 lên 1).\n\nChiến lược bổ sung:\n\n- Sử dụng PCIAT-PCIAT_Total như một biến mục tiêu liên tục, thực hiện hồi quy trên PCIAT-PCIAT_Total, sau đó ánh xạ kết quả dự đoán vào các mức sii tương ứng.\n- Dự đoán từng câu trả lời trong Bài kiểm tra Parent-Child Internet Addiction Test (PCIAT), sau đó cộng tổng các điểm dự đoán để tính PCIAT-PCIAT_Total và chuyển đổi sang danh mục sii phù hợp.","metadata":{}},{"cell_type":"markdown","source":"# Ngắm qua dữ liệu","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ndata_dict = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:17:11.683569Z","iopub.execute_input":"2024-11-30T02:17:11.684139Z","iopub.status.idle":"2024-11-30T02:17:11.728423Z","shell.execute_reply.started":"2024-11-30T02:17:11.684098Z","shell.execute_reply":"2024-11-30T02:17:11.727798Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Train data","metadata":{}},{"cell_type":"code","source":"display(train.head())\nprint(f\"Train shape: {train.shape}\")\ntrain.describe().transpose()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:17:11.730408Z","iopub.execute_input":"2024-11-30T02:17:11.730757Z","iopub.status.idle":"2024-11-30T02:17:11.898344Z","shell.execute_reply.started":"2024-11-30T02:17:11.730719Z","shell.execute_reply":"2024-11-30T02:17:11.897588Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Test data","metadata":{}},{"cell_type":"code","source":"display(test.head())\nprint(f\"Test shape: {test.shape}\")\ntest.describe().transpose()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:17:11.8993Z","iopub.execute_input":"2024-11-30T02:17:11.899578Z","iopub.status.idle":"2024-11-30T02:17:12.01665Z","shell.execute_reply.started":"2024-11-30T02:17:11.899553Z","shell.execute_reply":"2024-11-30T02:17:12.015879Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Data dictionary","metadata":{}},{"cell_type":"code","source":"data_dict.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:17:12.017899Z","iopub.execute_input":"2024-11-30T02:17:12.018568Z","iopub.status.idle":"2024-11-30T02:17:12.029203Z","shell.execute_reply.started":"2024-11-30T02:17:12.018517Z","shell.execute_reply":"2024-11-30T02:17:12.028312Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Ý nghĩa các cột bị thiếu trong test","metadata":{}},{"cell_type":"code","source":"train_cols = set(train.columns)\ntest_cols = set(test.columns)\ncolumns_not_in_test = sorted(list(train_cols - test_cols))\ndata_dict[data_dict['Field'].isin(columns_not_in_test)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:17:12.030546Z","iopub.execute_input":"2024-11-30T02:17:12.030792Z","iopub.status.idle":"2024-11-30T02:17:12.048571Z","shell.execute_reply.started":"2024-11-30T02:17:12.030767Z","shell.execute_reply":"2024-11-30T02:17:12.047786Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Kiểu dữ liệu của các field\ndef type_of_field(data):\n    \n    str_columns = data.select_dtypes(include=['object']).columns\n    int_columns = data.select_dtypes(include=['int64']).columns\n    float_columns = data.select_dtypes(include=['float64']).columns\n    other_columns = data.select_dtypes(exclude=['object', 'int64', 'float64']).columns\n    \n    # Tạo DataFrame để hiển thị dưới dạng bảng\n    data_types_df = pd.DataFrame({\n        'String Columns': pd.Series(str_columns),\n        'Integer Columns': pd.Series(int_columns),\n        'Float Columns': pd.Series(float_columns),\n        'Other Columns': pd.Series(other_columns)\n    })\n    return data_types_df\n\n# Hiển thị DataFrame\ndisplay(type_of_field(train))\ndisplay(type_of_field(test))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:17:12.049671Z","iopub.execute_input":"2024-11-30T02:17:12.049963Z","iopub.status.idle":"2024-11-30T02:17:12.072731Z","shell.execute_reply.started":"2024-11-30T02:17:12.049924Z","shell.execute_reply":"2024-11-30T02:17:12.072001Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Hàm hỗ trợ thống kê các giá trị trong cột","metadata":{}},{"cell_type":"code","source":"def calculate_stats(data, columns):\n    if isinstance(columns, str):\n        columns = [columns]\n\n    stats = []\n    for col in columns:\n        if data[col].dtype in ['object', 'category']:\n            counts = data[col].value_counts(dropna=False, sort=False)\n            percents = data[col].value_counts(normalize=True, dropna=False, sort=False) * 100\n            formatted = counts.astype(str) + ' (' + percents.round(2).astype(str) + '%)'\n            stats_col = pd.DataFrame({'count (%)': formatted})\n            stats.append(stats_col)\n        else:\n            stats_col = data[col].describe().to_frame().transpose()\n            stats_col['missing'] = data[col].isnull().sum()\n            stats_col.index.name = col\n            stats.append(stats_col)\n\n    return pd.concat(stats, axis=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:17:12.074832Z","iopub.execute_input":"2024-11-30T02:17:12.075052Z","iopub.status.idle":"2024-11-30T02:17:12.08096Z","shell.execute_reply.started":"2024-11-30T02:17:12.07503Z","shell.execute_reply":"2024-11-30T02:17:12.079979Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **XỬ LÝ NULL**","metadata":{}},{"cell_type":"code","source":"# Dùng để tính phần trăm các giá trị trong cột\n# sort_by_nan = True, sẽ sắp xếp các cột theo tỉ lệ thiếu dữ liệu\n\ndef calculate_missing_stats(data, sort_by_nan=True):\n    columns = data.columns \n\n    missing_percentages = {}  # Dictionary lưu tỷ lệ NaN\n\n    for col in columns:\n        if data[col].dtype in ['object', 'category']:\n            # Tính số lượng và phần trăm các giá trị trong cột\n            counts = data[col].value_counts(dropna=False, sort=False)\n            percents = data[col].value_counts(normalize=True, dropna=False, sort=False) * 100\n            # Không cần lưu bảng thống kê chi tiết cho các cột kiểu object\n        else:\n            # Tính thống kê mô tả cho các cột số\n            missing_count = data[col].isnull().sum()\n            missing_percentage = (missing_count / len(data)) * 100  # Phần trăm NaN\n            missing_percentages[col] = missing_percentage\n\n    # Sắp xếp các cột theo tỷ lệ NaN nếu sort_by_nan được bật\n    if sort_by_nan:\n        sorted_missing = pd.Series(missing_percentages).sort_values(ascending=False)\n\n        # Vẽ biểu đồ tỷ lệ NaN\n        plt.figure(figsize=(12, 8))\n        sorted_missing.plot(kind='bar', color='skyblue')\n        plt.title('Tỷ lệ giá trị bị thiếu (NaN) của các cột')\n        plt.xlabel('Cột')\n        plt.ylabel('Tỷ lệ NaN (%)')\n        plt.xticks(rotation=90, ha='right', fontsize=10)\n        plt.show()\n\n    return missing_percentages\n\n# Sử dụng hàm (ví dụ): nếu muốn sắp xếp theo tỷ lệ NaN\ncalculate_missing_stats(train)\ncalculate_missing_stats(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:17:12.082128Z","iopub.execute_input":"2024-11-30T02:17:12.082726Z","iopub.status.idle":"2024-11-30T02:17:13.734144Z","shell.execute_reply.started":"2024-11-30T02:17:12.082688Z","shell.execute_reply":"2024-11-30T02:17:13.733323Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Loại bỏ các hàng có giá trị sii = NaN","metadata":{}},{"cell_type":"code","source":"target = 'sii'\ntrain_cleaned = train.dropna(subset=[target])\ntrain_cleaned.head()\ntrain_cleaned.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:17:13.735134Z","iopub.execute_input":"2024-11-30T02:17:13.735375Z","iopub.status.idle":"2024-11-30T02:17:13.753484Z","shell.execute_reply.started":"2024-11-30T02:17:13.73535Z","shell.execute_reply":"2024-11-30T02:17:13.752487Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def fill_na(df, strategy='mean'):\n    # Kiểm tra chiến lược thay thế\n    if strategy not in ['mean', 'median', 'zero']:\n        raise ValueError(\"Chiến lược không hợp lệ. Chọn từ 'mean', 'median', hoặc 'zero'.\")\n    \n    # Tạo một bản sao của DataFrame\n    df_copy = df.copy()\n\n    # Bỏ cột 'id'\n    df_copy = df_copy.drop(columns=['id'], axis=1)\n    \n    # Danh sách các cột cần map\n    season_cols = [\n        'Basic_Demos-Enroll_Season', \n        'CGAS-Season', \n        'Physical-Season', \n        'FGC-Season', \n        'BIA-Season', \n        'PCIAT-Season', \n        'SDS-Season', \n        'PreInt_EduHx-Season',\n        'Fitness_Endurance-Season',\n        'PAQ_A-Season',\n        'PAQ_C-Season'\n    ]\n    \n    # Mapping giá trị mùa\n    season_mapping = {\n        'Spring': 0,\n        'Summer': 1,\n        'Fall': 2,\n        'Winter': 3\n    }\n    \n    # Áp dụng mapping cho các cột trong danh sách\n    df_copy[season_cols] = df_copy[season_cols].apply(lambda col: col.map(season_mapping))\n\n    # Kiểm tra và loại bỏ cột 'id' nếu nó tồn tại\n    if 'id' in df_copy.columns:\n        df_copy = df_copy.drop(columns=['id'])\n    # Lặp qua từng cột để thay thế NaN\n    for col in df_copy.columns:\n        if strategy == 'mean':\n            # Tính trung bình của cột và thay thế NaN\n            mean_value = df_copy[col].mean()\n            df_copy[col].fillna(mean_value, inplace=True)\n        elif strategy == 'median':\n            # Tính trung vị của cột và thay thế NaN\n            median_value = df_copy[col].median()\n            df_copy[col].fillna(median_value, inplace=True)\n        elif strategy == 'zero':\n            # Thay thế NaN bằng 0\n            df_copy[col].fillna(0, inplace=True)\n\n    return df_copy","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:17:13.75465Z","iopub.execute_input":"2024-11-30T02:17:13.754928Z","iopub.status.idle":"2024-11-30T02:17:13.762149Z","shell.execute_reply.started":"2024-11-30T02:17:13.754902Z","shell.execute_reply":"2024-11-30T02:17:13.761325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_mean = fill_na(train_cleaned, 'mean')\ntrain_median = fill_na(train_cleaned, 'median')\ntrain_zero = fill_na(train_cleaned, 'zero')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:17:13.763408Z","iopub.execute_input":"2024-11-30T02:17:13.763756Z","iopub.status.idle":"2024-11-30T02:17:13.878779Z","shell.execute_reply.started":"2024-11-30T02:17:13.763718Z","shell.execute_reply":"2024-11-30T02:17:13.877966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_cleaned.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:17:13.879879Z","iopub.execute_input":"2024-11-30T02:17:13.880939Z","iopub.status.idle":"2024-11-30T02:17:13.936153Z","shell.execute_reply.started":"2024-11-30T02:17:13.8809Z","shell.execute_reply":"2024-11-30T02:17:13.935358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_mean.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:17:13.937223Z","iopub.execute_input":"2024-11-30T02:17:13.937514Z","iopub.status.idle":"2024-11-30T02:17:13.989197Z","shell.execute_reply.started":"2024-11-30T02:17:13.937489Z","shell.execute_reply":"2024-11-30T02:17:13.988387Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_median.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:17:13.99033Z","iopub.execute_input":"2024-11-30T02:17:13.990587Z","iopub.status.idle":"2024-11-30T02:17:14.054534Z","shell.execute_reply.started":"2024-11-30T02:17:13.99056Z","shell.execute_reply":"2024-11-30T02:17:14.053554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_zero.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:17:14.055765Z","iopub.execute_input":"2024-11-30T02:17:14.056109Z","iopub.status.idle":"2024-11-30T02:17:14.12113Z","shell.execute_reply.started":"2024-11-30T02:17:14.056076Z","shell.execute_reply":"2024-11-30T02:17:14.120211Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **XEM TƯƠNG QUAN DỮ LIỆU**","metadata":{}},{"cell_type":"markdown","source":"Tương quan các mùa với sii","metadata":{}},{"cell_type":"code","source":"# Danh sách các cột phân loại\ncategorical_columns = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n                       'FGC-Season', 'BIA-Season', 'PCIAT-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\n# Thiết lập kích thước cho đồ thị\nplt.figure(figsize=(16, 24))\n\n# Lặp qua từng cột phân loại và vẽ biểu đồ thanh\nfor i, col in enumerate(categorical_columns, 1):\n    plt.subplot(4, 2, i)  # 4 hàng, 2 cột, vẽ đồ thị thứ i\n    sns.barplot(x=col, y='sii', data=train_cleaned, ci=None)  # Vẽ biểu đồ thanh, không tính khoảng tin cậy\n    plt.xticks(rotation=45)  # Xoay nhãn trên trục x\n    plt.title(f\"'sii' vs {col}\")  # Tiêu đề cho biểu đồ\n\n# Điều chỉnh không gian giữa các biểu đồ\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:17:14.122202Z","iopub.execute_input":"2024-11-30T02:17:14.122473Z","iopub.status.idle":"2024-11-30T02:17:15.619498Z","shell.execute_reply.started":"2024-11-30T02:17:14.122446Z","shell.execute_reply":"2024-11-30T02:17:15.618656Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Tương quan của các cột có kiểu dữ liệu là số với sii","metadata":{}},{"cell_type":"code","source":"# Lọc các cột số (numerical columns)\nnumerical_cols = train_cleaned.select_dtypes(include=['float64', 'int64']).columns\n\n# Thiết lập số lượng biểu đồ trên mỗi hàng\nplots_per_row = 5\nn_rows = (len(numerical_cols) + plots_per_row - 1) // plots_per_row  # Tính số hàng\n\n# Thiết lập kích thước cho đồ thị\nplt.figure(figsize=(20, 4 * n_rows))\n\n# Lặp qua từng cột số và vẽ biểu đồ thanh\nfor i, col in enumerate(numerical_cols):\n    plt.subplot(n_rows, plots_per_row, i + 1)  # Vẽ biểu đồ thứ i\n    sns.barplot(x='sii', y=col, data=train_cleaned, ci=None)  # Biểu đồ thanh, không tính khoảng tin cậy\n    plt.title(col)  # Tiêu đề cho biểu đồ\n    plt.tight_layout()  # Điều chỉnh không gian giữa các biểu đồ\n\n# Hiển thị đồ thị\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:17:15.620742Z","iopub.execute_input":"2024-11-30T02:17:15.62111Z","iopub.status.idle":"2024-11-30T02:18:24.029627Z","shell.execute_reply.started":"2024-11-30T02:17:15.621069Z","shell.execute_reply":"2024-11-30T02:18:24.028774Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Chuyển Spring = 0, Summer = 1, Fall = 2, Winter = 3","metadata":{}},{"cell_type":"code","source":"season_cols = [\n    'Basic_Demos-Enroll_Season', \n    'Demos-Enroll_Season',\n    'CGAS-Season', \n    'Physical-Season', \n    'FGC-Season', \n    'BIA-Season', \n    'PCIAT-Season', \n    'SDS-Season', \n    'PreInt_EduHx-Season',\n    'Fitness_Endurance-Season',\n    'PAQ_A-Season',\n    'PAQ_C-Season'\n]\n\nseason_mapping = {\n    'Spring': 0,\n    'Summer': 1,\n    'Fall': 2,\n    'Winter': 3\n}\n\nfor col in season_cols:\n    if col in train_cleaned.columns:\n        train_cleaned.loc[:, col] = train_cleaned[col].replace(season_mapping)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:18:24.030785Z","iopub.execute_input":"2024-11-30T02:18:24.031126Z","iopub.status.idle":"2024-11-30T02:18:24.066135Z","shell.execute_reply.started":"2024-11-30T02:18:24.03109Z","shell.execute_reply":"2024-11-30T02:18:24.065533Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Bỏ cột id\ntrain_data_no_id = train_cleaned.drop(columns=['id'], errors='ignore')\n# print(train_data_no_id.dtypes)\n# Kiểm tra các giá trị không phải là số trong DataFrame\n# non_numeric_cols = train_data_no_id.select_dtypes(exclude=['number']).columns\n# for col in non_numeric_cols:\n#     print(f\"Unique values in {col}: {train_data_no_id[col].unique()}\")\n\n# Tính ma trận tương quan\ncorrelation_matrix = train_data_no_id.corr()\n\nplt.figure(figsize=(30, 30))\nsns.heatmap(correlation_matrix, annot=True, fmt='.1f', cmap='coolwarm', square=True)\nplt.title('Correlation Heatmap')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:18:24.067191Z","iopub.execute_input":"2024-11-30T02:18:24.067529Z","iopub.status.idle":"2024-11-30T02:18:35.83Z","shell.execute_reply.started":"2024-11-30T02:18:24.067499Z","shell.execute_reply":"2024-11-30T02:18:35.828995Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Độ tương quan với sii","metadata":{}},{"cell_type":"code","source":"# Sắp xếp giá trị tương quan với 'sii' theo thứ tự giảm dần\ncorrelation_with_sii = correlation_matrix['sii'].sort_values(ascending=False)\n\n# Tạo biểu đồ thanh\nplt.figure(figsize=(14, 10))\nsns.barplot(x=correlation_with_sii.index, y=correlation_with_sii.values, palette='viridis')\n\n# Thiết lập tiêu đề và nhãn cho các trục\nplt.title('Correlation with SII', fontsize=14)\nplt.xlabel('Features', fontsize=14)\nplt.ylabel('Correlation Coefficient', fontsize=14)\n\n# Xoay các nhãn trên trục x cho dễ đọc\nplt.xticks(rotation=45, ha='right', fontsize=10)\n\n# Hiển thị biểu đồ\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:18:35.831219Z","iopub.execute_input":"2024-11-30T02:18:35.831516Z","iopub.status.idle":"2024-11-30T02:18:37.279732Z","shell.execute_reply.started":"2024-11-30T02:18:35.831487Z","shell.execute_reply":"2024-11-30T02:18:37.278861Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **TỔNG HỢP PHÂN BỐ DỮ LIỆU**","metadata":{}},{"cell_type":"markdown","source":"# phân phối của các cột trong train ","metadata":{}},{"cell_type":"code","source":"for column in train.columns:\n    if column == 'id':\n        continue\n    \n    plt.figure(figsize=(10, 6))\n    \n    if train[column].dtype in ['int64', 'float64']:\n        sns.histplot(train[column], kde=True, bins=30, color='blue')\n        plt.title(f\"Distribution of {column}\", fontsize=16)\n        plt.xlabel(column, fontsize=12)\n        plt.ylabel(\"Frequency\", fontsize=12)\n    else:\n        sns.countplot(y=train[column], palette=\"viridis\", order=train[column].value_counts().index)\n        plt.title(f\"Value Counts of {column}\", fontsize=16)\n        plt.xlabel(\"Count\", fontsize=12)\n        plt.ylabel(column, fontsize=12)\n    \n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:18:37.283318Z","iopub.execute_input":"2024-11-30T02:18:37.283595Z","iopub.status.idle":"2024-11-30T02:19:12.406822Z","shell.execute_reply.started":"2024-11-30T02:18:37.283567Z","shell.execute_reply":"2024-11-30T02:19:12.405794Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Mối quan hệ giữa các đặc trưng và mục tiêu sii","metadata":{}},{"cell_type":"code","source":"data = train\n\nfor column in train.columns:\n    if column in ['sii', 'id']:\n        continue\n    \n    feature_to_plot = column\n    \n    if data[feature_to_plot].dtype in ['float64', 'int64']:\n        plt.figure(figsize=(10, 6))\n        sns.scatterplot(data=data, x=feature_to_plot, y=\"sii\", hue=\"sii\", palette=\"viridis\", alpha=0.8)\n        plt.title(f\"Scatter Plot of {feature_to_plot} vs SII\", fontsize=16)\n        plt.xlabel(feature_to_plot, fontsize=12)\n        plt.ylabel(\"SII (Label)\", fontsize=12)\n        plt.grid(True)\n        plt.show()\n    \n    else:\n        plt.figure(figsize=(10, 6))\n        sns.countplot(data=data, x=feature_to_plot, hue=\"sii\", palette=\"bright\")\n        plt.title(f\"Bar Plot of {feature_to_plot} vs SII\", fontsize=16)\n        plt.xlabel(feature_to_plot, fontsize=12)\n        plt.ylabel(\"Count\", fontsize=12)\n        plt.xticks(rotation=45)\n        plt.grid(axis='y')\n        plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:19:12.408149Z","iopub.execute_input":"2024-11-30T02:19:12.408604Z","iopub.status.idle":"2024-11-30T02:19:47.681819Z","shell.execute_reply.started":"2024-11-30T02:19:12.408544Z","shell.execute_reply":"2024-11-30T02:19:47.680979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Actigraphy","metadata":{}},{"cell_type":"code","source":"train_pl = pl.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\nactigraphy = pl.read_parquet('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id=0417c91e/part-0.parquet')\nactigraphy","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:19:47.683172Z","iopub.execute_input":"2024-11-30T02:19:47.683561Z","iopub.status.idle":"2024-11-30T02:19:47.720143Z","shell.execute_reply.started":"2024-11-30T02:19:47.683512Z","shell.execute_reply":"2024-11-30T02:19:47.719363Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n* step: Đơn vị bước (có thể là chỉ số mẫu hoặc thời gian trong chuỗi thời gian).\n* X, Y, Z: Dữ liệu về chuyển động trên ba trục không gian. Đây là các giá trị chuyển động thu được từ cảm biến (thường là gia tốc kế).\n* enmo: ENMO (Energy Expenditure) – Mức độ năng lượng tiêu thụ dựa trên chuyển động, được tính từ ba trục chuyển động X, Y, Z.\n* anglez: Góc quay của đối tượng, có thể liên quan đến vị trí của thiết bị.\n* non-wear_flag: Cờ đánh dấu khi thiết bị không được đeo (0: đeo, 1: không đeo).\n* light: Đo lường mức độ ánh sáng trong môi trường của đối tượng (có thể liên quan đến môi trường ánh sáng mà đối tượng tiếp xúc).\n* battery_voltage: Điện áp của pin thiết bị.\n* time_of_day: Thời gian trong ngày, có thể được đo bằng giây hoặc micro giây kể từ một mốc thời gian (ví dụ, thời gian Unix).\n* weekday: Ngày trong tuần (1 = Thứ Hai, 2 = Thứ Ba, ...).\n* quarter: Quý trong năm (1, 2, 3, 4).\n* relative_date_PCIAT: Ngày và thời gian tuyệt đối của sự kiện, tính từ một mốc thời gian ban đầu.\n","metadata":{}},{"cell_type":"code","source":"def analyze_actigraphy(id, only_one_week=False, small=False):\n    actigraphy = pl.read_parquet(f'/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id={id}/part-0.parquet')\n    day = actigraphy.get_column('relative_date_PCIAT') + actigraphy.get_column('time_of_day') / 86400e9\n    sample = train_pl.filter(pl.col('id') == id)\n    age = sample.get_column('Basic_Demos-Age').item()\n    sex = ['boy', 'girl'][sample.get_column('Basic_Demos-Sex').item()]\n    actigraphy = (\n        actigraphy\n        .with_columns(\n            (day.diff() * 86400).alias('diff_seconds'),\n            (np.sqrt(np.square(pl.col('X')) + np.square(pl.col('Y')) + np.square(pl.col('Z'))).alias('norm'))\n        )\n    )\n\n    if only_one_week:\n        start = np.ceil(day.min())\n        mask = (start <= day.to_numpy()) & (day.to_numpy() <= start + 7*3)\n        mask &= ~ actigraphy.get_column('non-wear_flag').cast(bool).to_numpy()\n    else:\n        mask = np.full(len(day), True)\n        \n    if small:\n        timelines = [\n            ('enmo', 'forestgreen'),\n            ('light', 'orange'),\n        ]\n    else:\n        timelines = [\n            ('X', 'm'),\n            ('Y', 'm'),\n            ('Z', 'm'),\n#             ('norm', 'c'),\n            ('enmo', 'forestgreen'),\n            ('anglez', 'lightblue'),\n            ('light', 'orange'),\n            ('non-wear_flag', 'chocolate')\n    #         ('diff_seconds', 'k'),\n        ]\n        \n    _, axs = plt.subplots(len(timelines), 1, sharex=True, figsize=(12, len(timelines) * 1.1 + 0.5))\n    for ax, (feature, color) in zip(axs, timelines):\n        ax.set_facecolor('#eeeeee')\n        ax.scatter(day.to_numpy()[mask],\n                   actigraphy.get_column(feature).to_numpy()[mask],\n                   color=color, label=feature, s=1)\n        ax.legend(loc='upper left', facecolor='#eeeeee')\n        if feature == 'diff_seconds':\n            ax.set_ylim(-0.5, 20.5)\n    axs[-1].set_xlabel('day')\n    axs[-1].xaxis.set_major_locator(MaxNLocator(integer=True))\n    plt.tight_layout()\n    axs[0].set_title(f'id={id}, {sex}, age={age}')\n    plt.show()\n\nanalyze_actigraphy('0417c91e', only_one_week=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:19:47.721336Z","iopub.execute_input":"2024-11-30T02:19:47.722084Z","iopub.status.idle":"2024-11-30T02:19:49.705966Z","shell.execute_reply.started":"2024-11-30T02:19:47.722042Z","shell.execute_reply":"2024-11-30T02:19:49.705084Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# TRAIN","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom sklearn.preprocessing import StandardScaler\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\n\n# Hàm xử lý file parquet\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\n# Hàm load dữ liệu time series\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n\n# Định nghĩa mô hình AutoEncoder\nclass AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim*3),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*3, encoding_dim*2),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*2, encoding_dim),\n            nn.ReLU()\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim*2),\n            nn.ReLU(),\n            nn.Linear(input_dim*2, input_dim*3),\n            nn.ReLU(),\n            nn.Linear(input_dim*3, input_dim),\n            nn.Sigmoid()\n        )\n        \n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded\n\n# Hàm huấn luyện AutoEncoder\ndef perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    \n    data_tensor = torch.FloatTensor(df_scaled)\n    \n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n    \n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(autoencoder.parameters())\n    \n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n            \n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}')\n                 \n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n        \n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n    \n    return df_encoded\n\n# Hàm Feature Engineering\ndef feature_engineering(df):\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1) \n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    \n    return df\n\n# Load dữ liệu\ntrain_cleaned = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nif 'sii' in train_cleaned.columns:\n    train_cleaned = train_cleaned[featuresCols + ['sii']].dropna(subset=['sii'])\nelse:\n    print(\"Cột 'sii' không tồn tại trong DataFrame. Vui lòng kiểm tra lại các bước xử lý dữ liệu.\")\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ndf_train_cleaned = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\n\ntrain_ts_encoded = perform_autoencoder(df_train_cleaned, encoding_dim=60, epochs=100, batch_size=32)\ntest_ts_encoded = perform_autoencoder(df_test, encoding_dim=60, epochs=100, batch_size=32)\n\ntrain_ts_encoded[\"id\"] = train_ts[\"id\"]\ntest_ts_encoded['id'] = test_ts[\"id\"]\n\ntrain_cleaned = pd.merge(train_cleaned, train_ts_encoded, how=\"left\", on='id')\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')\n\n# Điền khuyết dữ liệu bằng giá trị mean\ntrain_cleaned = train_cleaned.apply(lambda x: x.fillna(x.mean()) if x.dtype.kind in 'biufc' else x)\ntest = test.apply(lambda x: x.fillna(x.mean()) if x.dtype.kind in 'biufc' else x)\n\n\n# Áp dụng Feature Engineering\ntrain_cleaned = feature_engineering(train_cleaned)\ntest = feature_engineering(test)\n\n# Chuẩn bị dữ liệu đầu vào\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-CGAS_Score', 'Physical-BMI']  # Thêm các cột cần thiết\ntrain_cleaned = train_cleaned[featuresCols].dropna(subset=['sii'])\ntest = test[featuresCols]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:30:07.57918Z","iopub.execute_input":"2024-11-30T02:30:07.579595Z","iopub.status.idle":"2024-11-30T02:31:29.592864Z","shell.execute_reply.started":"2024-11-30T02:30:07.579562Z","shell.execute_reply":"2024-11-30T02:31:29.591582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 42\nn_splits = 5\n!pip -q install /kaggle/input/pytorchtabnet/pytorch_tabnet-4.1.0-py3-none-any.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:25:48.400111Z","iopub.status.idle":"2024-11-30T02:25:48.400688Z","shell.execute_reply.started":"2024-11-30T02:25:48.400436Z","shell.execute_reply":"2024-11-30T02:25:48.40046Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pytorch_tabnet.tab_model import TabNetRegressor\nimport torch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:25:48.402017Z","iopub.status.idle":"2024-11-30T02:25:48.402448Z","shell.execute_reply.started":"2024-11-30T02:25:48.402224Z","shell.execute_reply":"2024-11-30T02:25:48.402246Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train_cleaned = train.copy()\n# season_cols = [\n#     'Basic_Demos-Enroll_Season', \n#     'Demos-Enroll_Season',\n#     'CGAS-Season', \n#     'Physical-Season', \n#     'FGC-Season', \n#     'BIA-Season', \n#     'PCIAT-Season', \n#     'SDS-Season', \n#     'PreInt_EduHx-Season',\n#     'Fitness_Endurance-Season',\n#     'PAQ_A-Season',\n#     'PAQ_C-Season'\n# ]\n    \n# season_mapping = {\n#         'Spring': 0,\n#         'Summer': 1,\n#         'Fall': 2,\n#         'Winter': 3\n# }\n# target = 'sii'\n# train_cleaned = train.dropna(subset=[target])\n# # Áp dụng mapping cho các cột season\n# for col in season_cols:\n#     if col in train_cleaned.columns:\n#          train_cleaned[col] = train_cleaned[col].map(season_mapping)\n#     SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n# train_cleaned = train_cleaned.drop(columns=['id'])\n# train_cleaned.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:25:48.403579Z","iopub.status.idle":"2024-11-30T02:25:48.404038Z","shell.execute_reply.started":"2024-11-30T02:25:48.403795Z","shell.execute_reply":"2024-11-30T02:25:48.403834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def fill_na(df, strategy='mean'):\n#     # Kiểm tra chiến lược thay thế\n#     if strategy not in ['mean', 'median', 'zero']:\n#         raise ValueError(\"Chiến lược không hợp lệ. Chọn từ 'mean', 'median', hoặc 'zero'.\")\n    \n#     # Tạo một bản sao của DataFrame\n#     df_copy = df.copy()\n\n#     # Kiểm tra và loại bỏ cột 'id' nếu nó tồn tại\n#     if 'id' in df_copy.columns:\n#         df_copy = df_copy.drop(columns=['id'])\n#     # Lặp qua từng cột để thay thế NaN\n#     for col in df_copy.columns:\n#         if strategy == 'mean':\n#             # Tính trung bình của cột và thay thế NaN\n#             mean_value = df_copy[col].mean()\n#             df_copy[col].fillna(mean_value, inplace=True)\n#         elif strategy == 'median':\n#             # Tính trung vị của cột và thay thế NaN\n#             median_value = df_copy[col].median()\n#             df_copy[col].fillna(median_value, inplace=True)\n#         elif strategy == 'zero':\n#             # Thay thế NaN bằng 0\n#             df_copy[col].fillna(0, inplace=True)\n\n#     return df_copy\n# train_mean = fill_na(train_cleaned, 'mean')\n# train_median = fill_na(train_cleaned, 'median')\n# train_zero = fill_na(train_cleaned, 'median')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:25:48.405213Z","iopub.status.idle":"2024-11-30T02:25:48.405658Z","shell.execute_reply.started":"2024-11-30T02:25:48.405422Z","shell.execute_reply":"2024-11-30T02:25:48.405446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def TrainML(model_class, test_data):\n\n    X = train_cleaned.drop(['sii'], axis=1)\n    y = train_cleaned['sii']\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:25:48.407009Z","iopub.status.idle":"2024-11-30T02:25:48.407471Z","shell.execute_reply.started":"2024-11-30T02:25:48.407213Z","shell.execute_reply":"2024-11-30T02:25:48.40724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Params = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01,  # Increased from 2.68e-06\n    'device': 'gpu'\n\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED,\n    'tree_method': 'gpu_hist',\n\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10,  # Increase this value\n    'task_type': 'GPU'\n\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:25:48.408962Z","iopub.status.idle":"2024-11-30T02:25:48.409398Z","shell.execute_reply.started":"2024-11-30T02:25:48.40917Z","shell.execute_reply":"2024-11-30T02:25:48.409193Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import train_test_split\nfrom pytorch_tabnet.callbacks import Callback\nimport os\nimport torch\nfrom pytorch_tabnet.callbacks import Callback\n\nclass TabNetWrapper(BaseEstimator, RegressorMixin):\n    def __init__(self, **kwargs):\n        self.model = TabNetRegressor(**kwargs)\n        self.kwargs = kwargs\n        self.imputer = SimpleImputer(strategy='median')\n        self.best_model_path = 'best_tabnet_model.pt'\n        \n    def fit(self, X, y):\n        # Handle missing values\n        X_imputed = self.imputer.fit_transform(X)\n        \n        if hasattr(y, 'values'):\n            y = y.values\n            \n        # Create internal validation set\n        X_train, X_valid, y_train, y_valid = train_test_split(\n            X_imputed, \n            y, \n            test_size=0.2,\n            random_state=42\n        )\n        \n        # Train TabNet model\n        history = self.model.fit(\n            X_train=X_train,\n            y_train=y_train.reshape(-1, 1),\n            eval_set=[(X_valid, y_valid.reshape(-1, 1))],\n            eval_name=['valid'],\n            eval_metric=['mse'],\n            max_epochs=500,\n            patience=50,\n            batch_size=1024,\n            virtual_batch_size=128,\n            num_workers=0,\n            drop_last=False,\n            callbacks=[\n                TabNetPretrainedModelCheckpoint(\n                    filepath=self.best_model_path,\n                    monitor='valid_mse',\n                    mode='min',\n                    save_best_only=True,\n                    verbose=True\n                )\n            ]\n        )\n        \n        # Load the best model\n        if os.path.exists(self.best_model_path):\n            self.model.load_model(self.best_model_path)\n            os.remove(self.best_model_path)  # Remove temporary file\n        \n        return self\n    \n    def predict(self, X):\n        X_imputed = self.imputer.transform(X)\n        return self.model.predict(X_imputed).flatten()\n    \n    def __deepcopy__(self, memo):\n        # Add deepcopy support for scikit-learn\n        cls = self.__class__\n        result = cls.__new__(cls)\n        memo[id(self)] = result\n        for k, v in self.__dict__.items():\n            setattr(result, k, deepcopy(v, memo))\n        return result\n\n# TabNet hyperparameters\nTabNet_Params = {\n    'n_d': 64,              # Width of the decision prediction layer\n    'n_a': 64,              # Width of the attention embedding for each step\n    'n_steps': 5,           # Number of steps in the architecture\n    'gamma': 1.5,           # Coefficient for feature selection regularization\n    'n_independent': 2,     # Number of independent GLU layer in each GLU block\n    'n_shared': 2,          # Number of shared GLU layer in each GLU block\n    'lambda_sparse': 1e-4,  # Sparsity regularization\n    'optimizer_fn': torch.optim.Adam,\n    'optimizer_params': dict(lr=2e-2, weight_decay=1e-5),\n    'mask_type': 'entmax',\n    'scheduler_params': dict(mode=\"min\", patience=10, min_lr=1e-5, factor=0.5),\n    'scheduler_fn': torch.optim.lr_scheduler.ReduceLROnPlateau,\n    'verbose': 1,\n    'device_name': 'cuda' if torch.cuda.is_available() else 'cpu'\n}\n\nclass TabNetPretrainedModelCheckpoint(Callback):\n    def __init__(self, filepath, monitor='val_loss', mode='min', \n                 save_best_only=True, verbose=1):\n        super().__init__()  # Initialize parent class\n        self.filepath = filepath\n        self.monitor = monitor\n        self.mode = mode\n        self.save_best_only = save_best_only\n        self.verbose = verbose\n        self.best = float('inf') if mode == 'min' else -float('inf')\n        \n    def on_train_begin(self, logs=None):\n        self.model = self.trainer  # Use trainer itself as model\n        \n    def on_epoch_end(self, epoch, logs=None):\n        logs = logs or {}\n        current = logs.get(self.monitor)\n        if current is None:\n            return\n        \n        # Check if current metric is better than best\n        if (self.mode == 'min' and current < self.best) or \\\n           (self.mode == 'max' and current > self.best):\n            if self.verbose:\n                print(f'\\nEpoch {epoch}: {self.monitor} improved from {self.best:.4f} to {current:.4f}')\n            self.best = current\n            if self.save_best_only:\n                self.model.save_model(self.filepath)  # Save the entire model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:25:48.410882Z","iopub.status.idle":"2024-11-30T02:25:48.411326Z","shell.execute_reply.started":"2024-11-30T02:25:48.411096Z","shell.execute_reply":"2024-11-30T02:25:48.41112Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if np.any(np.isinf(train_cleaned)):\n    train_cleaned = train_cleaned.replace([np.inf, -np.inf], np.nan)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:25:48.412708Z","iopub.status.idle":"2024-11-30T02:25:48.41316Z","shell.execute_reply.started":"2024-11-30T02:25:48.412932Z","shell.execute_reply":"2024-11-30T02:25:48.412954Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\nTabNet_Model = TabNetWrapper(**TabNet_Params) # New","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:25:48.414157Z","iopub.status.idle":"2024-11-30T02:25:48.414578Z","shell.execute_reply.started":"2024-11-30T02:25:48.41436Z","shell.execute_reply":"2024-11-30T02:25:48.414381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"voting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    ('tabnet', TabNet_Model)\n])\n\nSubmission1 = TrainML(voting_model, test)\n\nSubmission1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:25:48.415505Z","iopub.status.idle":"2024-11-30T02:25:48.415951Z","shell.execute_reply.started":"2024-11-30T02:25:48.415707Z","shell.execute_reply":"2024-11-30T02:25:48.41573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n        \ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission\n\n# Model parameters for LightGBM\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01  # Increased from 2.68e-06\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'cat_features': cat_c,\n    'verbose': 0,\n    'l2_leaf_reg': 10  # Increase this value\n}\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\nTabNet_Model = TabNetWrapper(**TabNet_Params)  # New:TAbNet\n\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    ('tabnet', TabNet_Model)  # New:TabNet\n])\n\n# Train the ensemble model\nSubmission2 = TrainML(voting_model, test)\n\n# Save submission\n#Submission2.to_csv('submission.csv', index=False)\nSubmission2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:25:48.417548Z","iopub.status.idle":"2024-11-30T02:25:48.418013Z","shell.execute_reply.started":"2024-11-30T02:25:48.41776Z","shell.execute_reply":"2024-11-30T02:25:48.417783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tp_rounded = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n    return tp_rounded\n\nimputer = SimpleImputer(strategy='median')\n\nensemble = VotingRegressor(estimators=[\n    ('lgb', Pipeline(steps=[('imputer', imputer), ('regressor', LGBMRegressor(random_state=SEED))])),\n    ('xgb', Pipeline(steps=[('imputer', imputer), ('regressor', XGBRegressor(random_state=SEED))])),\n    ('cat', Pipeline(steps=[('imputer', imputer), ('regressor', CatBoostRegressor(random_state=SEED, silent=True))])),\n    ('rf', Pipeline(steps=[('imputer', imputer), ('regressor', RandomForestRegressor(random_state=SEED))])),\n    ('gb', Pipeline(steps=[('imputer', imputer), ('regressor', GradientBoostingRegressor(random_state=SEED))])),\n    ('tabnet', Pipeline(steps=[('imputer', imputer), ('regressor', TabNetWrapper(**TabNet_Params))]))  # New:TabNet\n])\n\nSubmission3 = TrainML(ensemble, test)\n\nSubmission3 = TrainML(ensemble, test)\nSubmission3 = pd.DataFrame({\n    'id': sample['id'],\n    'sii': Submission3\n})\n\nSubmission3","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:25:48.419451Z","iopub.status.idle":"2024-11-30T02:25:48.419897Z","shell.execute_reply.started":"2024-11-30T02:25:48.419657Z","shell.execute_reply":"2024-11-30T02:25:48.419679Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub1 = Submission1\nsub2 = Submission2\nsub3 = Submission3\n\nsub1 = sub1.sort_values(by='id').reset_index(drop=True)\nsub2 = sub2.sort_values(by='id').reset_index(drop=True)\nsub3 = sub3.sort_values(by='id').reset_index(drop=True)\n\ncombined = pd.DataFrame({\n    'id': sub1['id'],\n    'sii_1': sub1['sii'],\n    'sii_2': sub2['sii'],\n    'sii_3': sub3['sii']\n})\n\ndef majority_vote(row):\n    return row.mode()[0]\n\ncombined['final_sii'] = combined[['sii_1', 'sii_2', 'sii_3']].apply(majority_vote, axis=1)\n\nfinal_submission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\n\nfinal_submission.to_csv('submission.csv', index=False)\n\nprint(\"Majority voting completed and saved to 'Final_Submission.csv'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:25:48.42147Z","iopub.status.idle":"2024-11-30T02:25:48.42193Z","shell.execute_reply.started":"2024-11-30T02:25:48.42168Z","shell.execute_reply":"2024-11-30T02:25:48.421702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T02:25:48.422983Z","iopub.status.idle":"2024-11-30T02:25:48.423426Z","shell.execute_reply.started":"2024-11-30T02:25:48.42319Z","shell.execute_reply":"2024-11-30T02:25:48.423211Z"}},"outputs":[],"execution_count":null}]}