{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.14"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":7453542,"sourceType":"datasetVersion","datasetId":921302},{"sourceId":203983,"sourceType":"modelInstanceVersion","modelInstanceId":172924,"modelId":195260}],"dockerImageVersionId":30762,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":673.251127,"end_time":"2024-12-12T20:25:29.287784","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-12-12T20:14:16.036657","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"The baseline was taken from [CMI | Reproducible results |FixSeed,LGB-CPU|LB.494](https://www.kaggle.com/code/kuosys/cmi-reproducible-results-fixseed-lgb-cpu-lb-494) 🙏","metadata":{"papermill":{"duration":0.007407,"end_time":"2024-12-12T20:14:18.711807","exception":false,"start_time":"2024-12-12T20:14:18.7044","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Metric Exploration","metadata":{"papermill":{"duration":0.00634,"end_time":"2024-12-12T20:14:18.724975","exception":false,"start_time":"2024-12-12T20:14:18.718635","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"Submissions are scored based on the quadratic weighted kappa, which measures the agreement between two outcomes. This metric typically varies from 0 (random agreement) to 1 (complete agreement). In the event that there is less agreement than expected by chance, the metric may go below 0.\n\nTo compute the quadratic weighted kappa, we construct three matrices, $O$, $W$, and $E$, with $N$ the number of distinct labels.\n\nThe matrix $O$ is an $N × 𝑁$ histogram matrix such that $O_{i,j}$ corresponds to the number of instances that have an actual value $i$ and a predicted value $j$.\n\nThe matrix $W$ is an $N × 𝑁$ matrix of weights, calculated based on the squared difference between actual and predicted values:\n\n$$ W_{i,j} = \\frac{(i - j)^2}{(N - 1)^2}.$$\n\nThe matrix $E$ is an $N × 𝑁$ histogram matrix of expected outcomes, calculated assuming that there is no correlation between values. This is calculated as the outer product between the actual histogram vector of outcomes and the predicted histogram vector, normalized such that $E$ and $O$ have the same sum.\n\nFrom these three matrices, the quadratic weighted kappa is calculated as: \n\n$$𝜅 = 1 - \\frac{\\sum_{i, j} W_{i, j} \\cdot O_{i, j}}{\\sum_{i, j} W_{i, j} \\cdot E_{i, j}}.$$\n\nThis metric can be interpreted as one minus the ratio of the squared error on the predicted values ​​to the error if we had randomly assigned predictions from a class distribution given by the proportions of predicted values. To learn more about this metric, you can find detailed information online, for example at [https://datatab.net/tutorial/weighted-cohens-kappa](https://datatab.net/tutorial/weighted-cohens-kappa).","metadata":{"papermill":{"duration":0.006269,"end_time":"2024-12-12T20:14:18.738524","exception":false,"start_time":"2024-12-12T20:14:18.732255","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nimport copy\nimport pickle\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport polars as pl\nimport polars.selectors as cs\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns\n\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\n\nimport plotly.express as px\n\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None","metadata":{"execution":{"iopub.status.busy":"2024-12-19T18:47:51.582971Z","iopub.status.idle":"2024-12-19T18:47:51.583394Z","shell.execute_reply.started":"2024-12-19T18:47:51.583165Z","shell.execute_reply":"2024-12-19T18:47:51.583186Z"},"papermill":{"duration":23.622575,"end_time":"2024-12-12T20:14:42.367516","exception":false,"start_time":"2024-12-12T20:14:18.744941","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n        \ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    print('OPTIMIZED THRESHOLDS', KappaOPtimizer.x)\n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n    optimized_thresholds = KappaOPtimizer.x\n    return submission, oof_tuned, oof_non_rounded, y, optimized_thresholds","metadata":{"execution":{"iopub.status.busy":"2024-12-19T18:47:51.585577Z","iopub.status.idle":"2024-12-19T18:47:51.586046Z","shell.execute_reply.started":"2024-12-19T18:47:51.585792Z","shell.execute_reply":"2024-12-19T18:47:51.585814Z"},"papermill":{"duration":73.652495,"end_time":"2024-12-12T20:15:56.027264","exception":false,"start_time":"2024-12-12T20:14:42.374769","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 42\nn_splits = 5\n\nmodel = XGBRegressor(\n    learning_rate=0.05,\n    max_depth=6,\n    n_estimators=200,\n    subsample=0.8,\n    colsample_bytree = 0.8,\n    reg_alpha=1,\n    reg_lambda=5,\n    random_state=SEED\n)\n\n# we get out of fold predictions for further exploration\nsubmission, y_pred, y_pred_non_rounded, y_true, optimized_thresholds = TrainML(model, test)","metadata":{"execution":{"iopub.execute_input":"2024-12-12T20:15:56.07009Z","iopub.status.busy":"2024-12-12T20:15:56.069819Z","iopub.status.idle":"2024-12-12T20:16:17.108687Z","shell.execute_reply":"2024-12-12T20:16:17.10776Z"},"papermill":{"duration":21.062077,"end_time":"2024-12-12T20:16:17.110347","exception":false,"start_time":"2024-12-12T20:15:56.04827","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Next, we simulate changes in the scores by adding one more observation (with classes 0, 1, 2, 3 specified by `y_new`) and calculate how much better/worse the metric becomes at different values of the predictions `pred_new` compared to the prediction 0.","metadata":{"papermill":{"duration":0.020364,"end_time":"2024-12-12T20:16:17.151975","exception":false,"start_time":"2024-12-12T20:16:17.131611","status":"completed"},"tags":[]}},{"cell_type":"code","source":"df_score_changes = []\nfor y_new in range(4):\n    item = {'y_new': y_new}\n    score_pred_zero = quadratic_weighted_kappa(list(y_true) + [y_new], list(y_pred) + [0])\n    for pred_new in range(4):\n        score = quadratic_weighted_kappa(list(y_true) + [y_new], list(y_pred) + [pred_new])\n        item[f'pred_new={pred_new}'] = score - score_pred_zero\n    df_score_changes.append(item)\n\ndf_score_changes = pd.DataFrame(df_score_changes)\ndf_score_changes","metadata":{"execution":{"iopub.execute_input":"2024-12-12T20:16:17.194494Z","iopub.status.busy":"2024-12-12T20:16:17.194211Z","iopub.status.idle":"2024-12-12T20:16:17.272418Z","shell.execute_reply":"2024-12-12T20:16:17.271617Z"},"papermill":{"duration":0.101721,"end_time":"2024-12-12T20:16:17.27411","exception":false,"start_time":"2024-12-12T20:16:17.172389","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Our analysis reveals that when `y_new=2`, predicting class 3 results in a higher score than predicting the true class 2 for a one addtional observation 🤯","metadata":{"papermill":{"duration":0.0204,"end_time":"2024-12-12T20:16:17.315784","exception":false,"start_time":"2024-12-12T20:16:17.295384","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"Let's plot the change in the metric as we vary the threshold (while keeping the others fixed).","metadata":{"papermill":{"duration":0.020529,"end_time":"2024-12-12T20:16:17.35683","exception":false,"start_time":"2024-12-12T20:16:17.336301","status":"completed"},"tags":[]}},{"cell_type":"code","source":"for t_idx in range(3):\n    df_plot = []\n    for t in np.arange(0.0, 3.0, 0.001):\n        thresholds = copy.copy(optimized_thresholds)\n        thresholds[t_idx] = t\n        score = -evaluate_predictions(thresholds, y_true, y_pred_non_rounded)\n        df_plot.append({f't_{t_idx}': t, 'score': score})\n    \n    df_plot = pd.DataFrame(df_plot)\n    fig = px.line(df_plot, x=f't_{t_idx}', y='score', title=f't_{t_idx}')\n    fig.show(renderer='iframe')","metadata":{"execution":{"iopub.execute_input":"2024-12-12T20:16:17.399059Z","iopub.status.busy":"2024-12-12T20:16:17.398803Z","iopub.status.idle":"2024-12-12T20:16:40.083006Z","shell.execute_reply":"2024-12-12T20:16:40.082116Z"},"papermill":{"duration":22.70751,"end_time":"2024-12-12T20:16:40.084859","exception":false,"start_time":"2024-12-12T20:16:17.377349","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# The threshold optimizer in the code appears to be finding a local maximum, not the global maximum\nprint('optimized_thresholds score:', -evaluate_predictions(optimized_thresholds, y_true, y_pred_non_rounded))\nprint('another thresholds score:', -evaluate_predictions([0.6264773 , 0.89171596, 1.64], y_true, y_pred_non_rounded))","metadata":{"execution":{"iopub.execute_input":"2024-12-12T20:16:40.129054Z","iopub.status.busy":"2024-12-12T20:16:40.128375Z","iopub.status.idle":"2024-12-12T20:16:40.138389Z","shell.execute_reply":"2024-12-12T20:16:40.137602Z"},"papermill":{"duration":0.033475,"end_time":"2024-12-12T20:16:40.139948","exception":false,"start_time":"2024-12-12T20:16:40.106473","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Issues with the Quadratic Weighted Kappa metric\n\n* **Non-intuitive behavior:** It's been shown that predicting incorrect values can sometimes result in a better QWK score than predicting the actual values. This highlights the complexity of interpreting changes in the metric and can make it difficult to understand whether model improvements are truly meaningful.\n* **Loss of information due to discretization:** QWK requires discrete predictions, which means continuous model outputs need to be thresholded into distinct categories. This process can obscure a significant amount of information about the model's performance. For example, QWK doesn't consider how well items are ranked within each predicted class. Items on the edge of thresholds and those deep inside a category are treated equally, potentially masking important distinctions. In my opinion, a continuous metric might provide a more nuanced and informative assessment of model quality, potentially leading to better business decisions.","metadata":{"papermill":{"duration":0.020504,"end_time":"2024-12-12T20:16:40.18136","exception":false,"start_time":"2024-12-12T20:16:40.160856","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# BASELINE","metadata":{"papermill":{"duration":0.020535,"end_time":"2024-12-12T20:16:40.222671","exception":false,"start_time":"2024-12-12T20:16:40.202136","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Issues with the following baseline (as well as in similar top-performing models on a public dataset)\n* Autoencoder for `train` and `test` dataset is fitted separately and it might lead to severely different encoding for the same data (see using `perform_autoencoder`)\n* Thresholds optimizer finds only local extremum (see using `KappaOPtimizer = minimize(...)`)\n\nIncredibly, this dubious technique gives a good score on the public leaderboard 🤷. Let's hope that for private dataset something more correct will work better 🙏.","metadata":{"papermill":{"duration":0.020695,"end_time":"2024-12-12T20:16:40.264012","exception":false,"start_time":"2024-12-12T20:16:40.243317","status":"completed"},"tags":[]}},{"cell_type":"code","source":"!pip -q install /kaggle/input/pytorchtabnet/pytorch_tabnet-4.1.0-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2024-12-19T18:51:55.166247Z","iopub.execute_input":"2024-12-19T18:51:55.166562Z","iopub.status.idle":"2024-12-19T18:52:36.834829Z","shell.execute_reply.started":"2024-12-19T18:51:55.166535Z","shell.execute_reply":"2024-12-19T18:52:36.833780Z"},"papermill":{"duration":40.674512,"end_time":"2024-12-12T20:17:20.959469","exception":false,"start_time":"2024-12-12T20:16:40.284957","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pytorch_tabnet.tab_model import TabNetRegressor\nimport torch","metadata":{"execution":{"iopub.status.busy":"2024-12-19T18:52:40.406003Z","iopub.execute_input":"2024-12-19T18:52:40.406764Z","iopub.status.idle":"2024-12-19T18:52:44.263381Z","shell.execute_reply.started":"2024-12-19T18:52:40.406728Z","shell.execute_reply":"2024-12-19T18:52:44.262358Z"},"papermill":{"duration":0.039431,"end_time":"2024-12-12T20:17:21.020376","exception":false,"start_time":"2024-12-12T20:17:20.980945","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport polars as pl\nimport polars.selectors as cs\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns\n\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None","metadata":{"execution":{"iopub.status.busy":"2024-12-19T18:52:46.597176Z","iopub.execute_input":"2024-12-19T18:52:46.597943Z","iopub.status.idle":"2024-12-19T18:53:01.010002Z","shell.execute_reply.started":"2024-12-19T18:52:46.597909Z","shell.execute_reply":"2024-12-19T18:53:01.009266Z"},"papermill":{"duration":0.029756,"end_time":"2024-12-12T20:17:21.071019","exception":false,"start_time":"2024-12-12T20:17:21.041263","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\ndef seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True\nseed_everything(2024)","metadata":{"execution":{"iopub.status.busy":"2024-12-19T18:53:02.430986Z","iopub.execute_input":"2024-12-19T18:53:02.432156Z","iopub.status.idle":"2024-12-19T18:53:02.443128Z","shell.execute_reply.started":"2024-12-19T18:53:02.432118Z","shell.execute_reply":"2024-12-19T18:53:02.441952Z"},"papermill":{"duration":0.03252,"end_time":"2024-12-12T20:17:21.125269","exception":false,"start_time":"2024-12-12T20:17:21.092749","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 42\nn_splits = 5","metadata":{"execution":{"iopub.status.busy":"2024-12-19T18:53:05.374052Z","iopub.execute_input":"2024-12-19T18:53:05.374730Z","iopub.status.idle":"2024-12-19T18:53:05.378713Z","shell.execute_reply.started":"2024-12-19T18:53:05.374696Z","shell.execute_reply":"2024-12-19T18:53:05.377792Z"},"papermill":{"duration":0.0261,"end_time":"2024-12-12T20:17:21.172107","exception":false,"start_time":"2024-12-12T20:17:21.146007","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Engineering\n\n- **Feature Selection**: The dataset contains features related to physical characteristics (e.g., BMI, Height, Weight), behavioral aspects (e.g., internet usage), and fitness data (e.g., endurance time). \n- **Categorical Feature Encoding**: Categorical features are mapped to numerical values using custom mappings for each unique category within the dataset. This ensures compatibility with machine learning algorithms that require numerical input.\n- **Time Series Aggregation**: Time series statistics (e.g., mean, standard deviation) from the actigraphy data are computed and merged into the main dataset to create additional features for model training.\n","metadata":{"papermill":{"duration":0.020578,"end_time":"2024-12-12T20:17:21.213535","exception":false,"start_time":"2024-12-12T20:17:21.192957","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    df = df.loc[:, ['X', 'Y', 'Z', 'enmo', 'anglez']]\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n\n\nclass AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim*3),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*3, encoding_dim*2),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*2, encoding_dim),\n            nn.ReLU()\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim*2),\n            nn.ReLU(),\n            nn.Linear(input_dim*2, input_dim*3),\n            nn.ReLU(),\n            nn.Linear(input_dim*3, input_dim),\n            nn.Sigmoid()\n        )\n        \n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded\n\n\ndef perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    \n    data_tensor = torch.FloatTensor(df_scaled)\n    \n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n    \n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(autoencoder.parameters())\n    \n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n            \n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n                 \n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n        \n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n    \n    return df_encoded\n\ndef feature_engineering(df):\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1) \n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    df['BMI_PHR'] = df['Physical-BMI'] * df['Physical-HeartRate']\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-12-19T18:53:07.638093Z","iopub.execute_input":"2024-12-19T18:53:07.638480Z","iopub.status.idle":"2024-12-19T18:53:07.655113Z","shell.execute_reply.started":"2024-12-19T18:53:07.638449Z","shell.execute_reply":"2024-12-19T18:53:07.653924Z"},"papermill":{"duration":0.037195,"end_time":"2024-12-12T20:17:21.271465","exception":false,"start_time":"2024-12-12T20:17:21.23427","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Tíme series feature engineering","metadata":{}},{"cell_type":"code","source":"import os \nfrom tqdm import tqdm \ndef extract_enmo(df_source, id=None):\n    df = df_source.copy()\n    df = df[df['non-wear_flag'] == 0]\n    df.drop('non-wear_flag', axis=1, inplace=True)\n    df.loc[:, 'Type_activity'] = 'Non-assigned'\n    df.loc[(df['enmo'] < 10*1e-3), 'Type_activity'] = 'sedentary'\n    df.loc[(df['enmo'] >= 10*1e-3) & (df['enmo'] < 100*1e-3), 'Type_activity'] = 'light'\n    df.loc[(df['enmo'] >= 100*1e-3), 'Type_activity'] = 'moderate'\n    \n    total_wear = df['step'].count()\n    \n    sedentary_perall = df[df['Type_activity'] == 'sedentary']['step'].count()\n    sedentary_perall = sedentary_perall / total_wear\n    \n    light_perall = df[df['Type_activity'] == 'light']['step'].count()\n    light_perall = light_perall / total_wear\n    \n    moderate_perall = df[df['Type_activity'] == 'moderate']['step'].count()\n    moderate_perall = moderate_perall / total_wear\n\n    sedentary_perall, light_perall, moderate_perall\n    return pd.DataFrame({'id': [id], \n                         'sedentary_por': [sedentary_perall], \n                         'light_por': [light_perall],\n                         'moderate_por': [moderate_perall]}\n                       )\n\ndef getEnmo(ts_path):\n    listdir = os.listdir(ts_path)\n    res_df = None\n    for dir in tqdm(listdir):\n        # print(dir)\n        dft = pd.read_parquet(os.path.join(ts_path, dir, \"part-0.parquet\"))\n        \n        id = dir[3:]\n        ex_df = extract_enmo(dft, id=id)\n        if res_df is None:\n            res_df = ex_df\n        else:\n            res_df = pd.concat([res_df, ex_df])\n    return res_df\n\n#res_df là kết quả","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:53:10.982059Z","iopub.execute_input":"2024-12-19T18:53:10.982967Z","iopub.status.idle":"2024-12-19T18:53:10.994609Z","shell.execute_reply.started":"2024-12-19T18:53:10.982920Z","shell.execute_reply":"2024-12-19T18:53:10.993762Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom concurrent.futures import ThreadPoolExecutor\nfrom torch.utils.data import DataLoader\nimport torch\nfrom torch import nn\nfrom torch.nn import functional as F\nfrom sklearn.preprocessing import StandardScaler\n\nclass CNN(nn.Module):\n    def __init__(self):\n        super(CNN, self).__init__()\n        self.cur_epoches = 0\n        #input (N, 5)\n        self.conv1 = nn.Conv1d(5, 64, kernel_size=3, stride=2, padding='valid') # 32, N, 1\n        self.avgpool1 = nn.MaxPool1d(kernel_size=2, stride=2)\n\n        self.conv2 = nn.Conv1d(64, 128, kernel_size=3, stride=2, padding='valid') # 32, N, 1\n        self.avgpool2 = nn.MaxPool1d(kernel_size=2, stride=2)\n\n        self.conv3 = nn.Conv1d(128, 128, kernel_size=3, stride=2, padding='valid') # 32, N, 1\n        self.avgpool3 = nn.MaxPool1d(kernel_size=2, stride=2)\n\n        self.conv4 = nn.Conv1d(128, 256, kernel_size=3, stride=2, padding='valid') # 32, N, 1\n        self.avgpool4 = nn.MaxPool1d(kernel_size=2, stride=2)\n\n        self.conv5 = nn.Conv1d(256, 256, kernel_size=3, stride=2, padding='valid')\n        \n        self.fc1 = nn.Linear(256, 128) # Adjust output size based on input dims\n        self.fc15 = nn.Linear(128, 64)\n        self.fc2 = nn.Linear(64, 4)\n\n        self.dropout1 = nn.Dropout(0.3, inplace=False)\n        self.dropout2 = nn.Dropout(0.3, inplace=False)\n\n        self.act = nn.LeakyReLU(0.1)\n        self.act1 = nn.Sigmoid()\n\n    def forward(self, x, debug=False):\n        x = self.act(self.conv1(x))\n        if debug: print(x.shape)\n        x = self.avgpool1(x)\n        if debug: print(x.shape)\n\n        x = self.act(self.conv2(x))\n        if debug: print(x.shape)\n        x = self.avgpool2(x)\n        if debug: print(x.shape)\n\n        x = self.act(self.conv3(x))\n        if debug: print(x.shape)\n        x = self.avgpool3(x)\n        if debug: print(x.shape)\n\n        x = self.act(self.conv4(x))\n        if debug: print(x.shape)\n        x = self.avgpool4(x)\n        if debug: print(x.shape)\n\n        x = self.act(self.conv5(x))\n        if debug: print(x.shape)\n\n        x = F.adaptive_avg_pool1d(x, 1).squeeze()\n        if debug: print(x.shape)\n\n        x = self.act1(self.fc1(x))\n        if debug: print(x.shape)\n        x = self.dropout1(x)\n        \n        x = self.act1(self.fc15(x))\n        x = self.dropout2(x)\n        \n        x = self.fc2(x)\n        if debug: print(x.shape)\n        x = F.softmax(x, dim=0)\n\n        return x\n\n    def feature_extract(self, x, debug=False):\n        x = self.act(self.conv1(x))\n        if debug: print(x.shape)\n        x = self.avgpool1(x)\n        if debug: print(x.shape)\n\n        x = self.act(self.conv2(x))\n        if debug: print(x.shape)\n        x = self.avgpool2(x)\n        if debug: print(x.shape)\n\n        x = self.act(self.conv3(x))\n        if debug: print(x.shape)\n        x = self.avgpool3(x)\n        if debug: print(x.shape)\n\n        x = self.act(self.conv4(x))\n        if debug: print(x.shape)\n        x = self.avgpool4(x)\n        if debug: print(x.shape)\n\n        x = self.act(self.conv5(x))\n        if debug: print(x.shape)\n\n        x = F.adaptive_avg_pool1d(x, 1).squeeze()\n        if debug: print(x.shape)\n\n        x = self.act1(self.fc1(x))\n        if debug: print(x.shape)\n        x = self.dropout1(x)\n        \n        x = self.act1(self.fc15(x)) #64\n        y = self.dropout2(x)\n        \n        y = self.fc2(y)\n        if debug: print(y.shape)\n        y = F.softmax(y, dim=0) #4\n\n        return x, y\n        ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:53:21.355240Z","iopub.execute_input":"2024-12-19T18:53:21.355595Z","iopub.status.idle":"2024-12-19T18:53:21.371439Z","shell.execute_reply.started":"2024-12-19T18:53:21.355564Z","shell.execute_reply":"2024-12-19T18:53:21.370304Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Dataset:\n    def __init__(self, device, path_tabu, path_ts, preload=True, type='train'):\n        self.device = device\n        self.path_ts = path_ts\n        self.tabu_data = pd.read_csv(path_tabu) \n        self.ids = [x[3:] for x in os.listdir(path_ts)]\n        self.filter()\n        self.ids.sort()\n        self.type = type\n        if type == 'train':\n            n = len(self.ids)\n            self.ids = self.ids[:int(n*0.8)]\n        elif type == 'val':\n            n = len(self.ids)\n            self.ids = self.ids[int(n*0.8):]\n        self.preload = preload\n        if self.preload:\n            self.ts_data_X, self.ts_data_Y = self.load_all_data()\n    def filter(self):\n        temp_ids = []\n        for id in tqdm(self.ids):\n            df = pd.read_parquet(os.path.join(self.path_ts, \"id=\" + id, \"part-0.parquet\"))\n            if df.shape[0] >= 900:\n                temp_ids.append(id)\n        self.ids = temp_ids\n\n    def collate(self, index):\n        X = [self.ts_data_X[i].to(self.device) for i in index]\n        Y = None\n        if self.ts_data_Y is not None:\n            Y = [self.ts_data_Y[i] for i in index]\n            Y = torch.tensor(Y, dtype=torch.int64)\n            Y = torch.nn.functional.one_hot(Y, num_classes=4).to(self.device).to(torch.float32)\n        return {'X': X,\n                'Y': Y}\n            \n    def dataloader(self, batch_size=1):\n        size = len(self.ts_data_X)\n        batch_size = size if batch_size == -1 else batch_size\n        loader = DataLoader(list(range(size)), batch_size=batch_size, collate_fn=self.collate, shuffle=True if self.type == 'test' else False, num_workers=0)\n        return loader\n        \n    def load_all_data(self):\n        count = 0\n        inputs = []\n        labels = None if self.type == 'test' else list()\n        for id in tqdm(self.ids):\n            df = pd.read_parquet(os.path.join(self.path_ts, \"id=\" + id, \"part-0.parquet\"))\n            df = df.loc[:, ['X', 'Y', 'Z', 'enmo', 'anglez']]\n            # normalize the signals \n            scaler = StandardScaler()\n            df = pd.DataFrame(scaler.fit_transform(df), columns=df.columns)\n\n            input = torch.tensor(df.values)\n            input = input.T\n            inputs.append(input)\n            if self.type != 'test': labels.append(self.tabu_data[self.tabu_data['id'] == id]['sii'].values[0])\n\n            \n            # count += 1\n            # if count == 2:\n            #     break\n        \n        return inputs, labels\n        \n# train_tabu_path=\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\"\n# train_ts_path='/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet'\n# dataset = Dataset('cpu', path_tabu=train_tabu_path, path_ts=train_ts_path, type='train')\n# dataloader = dataset.dataloader()\n# model.zero_grad()\n# for data in dataloader:\n#     print(data['X'][0])\n#     print(data['Y'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:53:24.377786Z","iopub.execute_input":"2024-12-19T18:53:24.378524Z","iopub.status.idle":"2024-12-19T18:53:24.390893Z","shell.execute_reply.started":"2024-12-19T18:53:24.378489Z","shell.execute_reply":"2024-12-19T18:53:24.389895Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_feature_64_4(tabu_path, ts_path, merge=False):\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    model_path = \"/kaggle/input/cnn-feature_extractor/pytorch/default/5/model-50 (1).pth\"\n    model = CNN()\n    model.load_state_dict(torch.load(model_path, weights_only=True))\n    model.to(device)\n    model.eval()\n\n    trainset = Dataset(device, path_tabu=tabu_path, path_ts=ts_path, type='test')\n    dataloader = trainset.dataloader(batch_size=1)\n\n\n    features64 = list()\n    features4 = list()\n    with torch.no_grad():\n        for data in dataloader:\n            X = data['X'][0]\n            x_64, x_4 = model.feature_extract(X)\n            features64.append(x_64)\n            features4.append(x_4)\n    \n    features64 = torch.stack(features64, dim=0)\n    features4 = torch.stack(features4, dim=0)\n    features64 = features64.cpu().detach().numpy()\n    features4 = features4.cpu().detach().numpy()\n    ids = np.array(trainset.ids)\n    if merge:\n        features68 = np.concatenate([features64, features4], axis=-1)\n        ts_features_df = pd.DataFrame(features68, columns=[f'feature_{i}' for i in range(features68.shape[1])])\n        ts_features_df.insert(0, 'id', ids)\n        return ts_features_df\n    else:\n        ts_features_df64 = pd.DataFrame(features64, columns=[f'feature64_{i}' for i in range(features64.shape[1])])\n        ts_features_df64.insert(0, 'id', ids)\n\n        ts_features_df4 = pd.DataFrame(features4, columns=[f'feature4_{i}' for i in range(features4.shape[1])])\n        ts_features_df4.insert(0, 'id', ids)\n        return ts_features_df64, ts_features_df4","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:53:27.212553Z","iopub.execute_input":"2024-12-19T18:53:27.212879Z","iopub.status.idle":"2024-12-19T18:53:27.221298Z","shell.execute_reply.started":"2024-12-19T18:53:27.212849Z","shell.execute_reply":"2024-12-19T18:53:27.220325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#tabu_path_train = '/kaggle/input/child-mind-institute-problematic-internet-use/train.csv'\n#ts_path_train = '/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet'\n\n#tabu_path_test = '/kaggle/input/child-mind-institute-problematic-internet-use/test.csv'\n#ts_path_test = '/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet'\n\n#train_68 = get_feature_64_4(tabu_path_train, ts_path_train, merge=True)\n#train_68","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T16:18:16.959120Z","iopub.execute_input":"2024-12-19T16:18:16.959732Z","iopub.status.idle":"2024-12-19T16:19:34.294744Z","shell.execute_reply.started":"2024-12-19T16:18:16.959698Z","shell.execute_reply":"2024-12-19T16:19:34.293928Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#l = train_64.columns.tolist()\n#l.remove('id')\n#print(l)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T17:01:14.966383Z","iopub.execute_input":"2024-12-19T17:01:14.966723Z","iopub.status.idle":"2024-12-19T17:01:14.971730Z","shell.execute_reply.started":"2024-12-19T17:01:14.966695Z","shell.execute_reply":"2024-12-19T17:01:14.970871Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#train_64, train_4 = get_feature_64_4(tabu_path_train, ts_path_train, merge=False)\n#train_64","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T16:19:49.622659Z","iopub.execute_input":"2024-12-19T16:19:49.622986Z","iopub.status.idle":"2024-12-19T16:21:05.391033Z","shell.execute_reply.started":"2024-12-19T16:19:49.622957Z","shell.execute_reply":"2024-12-19T16:21:05.390177Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#train_4","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T16:21:15.360965Z","iopub.execute_input":"2024-12-19T16:21:15.361756Z","iopub.status.idle":"2024-12-19T16:21:15.374854Z","shell.execute_reply.started":"2024-12-19T16:21:15.361720Z","shell.execute_reply":"2024-12-19T16:21:15.373667Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#train_enmo = getEnmo(ts_path_train)\n#test_enmo = getEnmo(ts_path_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T16:25:48.177425Z","iopub.execute_input":"2024-12-19T16:25:48.178285Z","iopub.status.idle":"2024-12-19T16:27:30.399417Z","shell.execute_reply.started":"2024-12-19T16:25:48.178250Z","shell.execute_reply":"2024-12-19T16:27:30.398218Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#test_enmo","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T16:27:41.736711Z","iopub.execute_input":"2024-12-19T16:27:41.737593Z","iopub.status.idle":"2024-12-19T16:27:41.746397Z","shell.execute_reply.started":"2024-12-19T16:27:41.737557Z","shell.execute_reply":"2024-12-19T16:27:41.745589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#train_enmo","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T16:33:13.100844Z","iopub.execute_input":"2024-12-19T16:33:13.101791Z","iopub.status.idle":"2024-12-19T16:33:13.115640Z","shell.execute_reply.started":"2024-12-19T16:33:13.101741Z","shell.execute_reply":"2024-12-19T16:33:13.114728Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# End","metadata":{}},{"cell_type":"code","source":"tabu_path_train = '/kaggle/input/child-mind-institute-problematic-internet-use/train.csv'\nts_path_train = '/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet'\n\ntabu_path_test = '/kaggle/input/child-mind-institute-problematic-internet-use/test.csv'\nts_path_test = '/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet'\n\ntrain_64, train_4 = get_feature_64_4(tabu_path_train, ts_path_train, merge=False)\ntest_64, test_4 = get_feature_64_4(tabu_path_test, ts_path_test, merge=False)\n\ntrain_enmo = getEnmo(ts_path_train)\ntest_enmo = getEnmo(ts_path_test)\n\ntrain_ts_7 = train_enmo.merge(train_4, on ='id', how='left')\ntest_ts_7 = test_enmo.merge(test_4, on='id', how='left')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:55:16.496930Z","iopub.execute_input":"2024-12-19T18:55:16.497754Z","iopub.status.idle":"2024-12-19T18:59:10.312522Z","shell.execute_reply.started":"2024-12-19T18:55:16.497717Z","shell.execute_reply":"2024-12-19T18:59:10.311458Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ndf_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\n\ntrain_ts_encoded = perform_autoencoder(df_train, encoding_dim=60, epochs=100, batch_size=32)\ntest_ts_encoded = perform_autoencoder(df_test, encoding_dim=60, epochs=100, batch_size=32)\n\ntrain_ts_encoded[\"id\"]=train_ts[\"id\"]\ntest_ts_encoded['id']=test_ts[\"id\"]\n\ntrain_ts_encoded = train_ts_encoded.merge(train_ts_7, on='id', how='left')\ntest_ts_encoded = test_ts_encoded.merge(test_ts_7, on='id', how='left')\n\ntime_series_cols = train_ts_encoded.columns.tolist()\ntime_series_cols.remove('id')\n\n\ntrain = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')\n\nimputer = KNNImputer(n_neighbors=5)\nnumeric_cols = train.select_dtypes(include=['float64', 'int64']).columns\nimputed_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\n        \ntrain = train_imputed\n\ntrain = feature_engineering(train)\ntrain = train.dropna(thresh=10, axis=0)\ntest = feature_engineering(test)\n\ntrain = train.drop('id', axis=1)\ntest  = test .drop('id', axis=1)   \n\n\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW','BMI_PHR']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW','BMI_PHR']\n\nfeaturesCols += time_series_cols\ntest = test[featuresCols]","metadata":{"execution":{"iopub.status.busy":"2024-12-19T18:59:12.944536Z","iopub.execute_input":"2024-12-19T18:59:12.944909Z","iopub.status.idle":"2024-12-19T19:00:49.613399Z","shell.execute_reply.started":"2024-12-19T18:59:12.944878Z","shell.execute_reply":"2024-12-19T19:00:49.612597Z"},"papermill":{"duration":86.896067,"end_time":"2024-12-12T20:18:48.188603","exception":false,"start_time":"2024-12-12T20:17:21.292536","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if np.any(np.isinf(train)):\n    train = train.replace([np.inf, -np.inf], np.nan)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)","metadata":{"execution":{"iopub.status.busy":"2024-12-19T19:00:57.337555Z","iopub.execute_input":"2024-12-19T19:00:57.338597Z","iopub.status.idle":"2024-12-19T19:00:57.349722Z","shell.execute_reply.started":"2024-12-19T19:00:57.338557Z","shell.execute_reply":"2024-12-19T19:00:57.348894Z"},"papermill":{"duration":0.048881,"end_time":"2024-12-12T20:18:48.276056","exception":false,"start_time":"2024-12-12T20:18:48.227175","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Training and Evaluation\n\n- **Model Types**: Various models are used, including:\n  - **LightGBM**: A gradient-boosting framework known for its speed and efficiency with large datasets.\n  - **XGBoost**: Another powerful gradient-boosting model used for structured data.\n  - **CatBoost**: Optimized for categorical features without the need for extensive preprocessing.\n  - **Voting Regressor**: An ensemble model that combines the predictions of LightGBM, XGBoost, and CatBoost for better accuracy.\n- **Cross-Validation**: Stratified K-Folds cross-validation is employed to split the data into training and validation sets, ensuring balanced class distribution in each fold.\n- **Quadratic Weighted Kappa (QWK)**: The performance of the models is evaluated using QWK, which measures the agreement between predicted and actual values, taking into account the ordinal nature of the target variable.\n- **Threshold Optimization**: The `minimize` function from `scipy.optimize` is used to fine-tune decision thresholds that map continuous predictions to discrete categories (None, Mild, Moderate, Severe).\n","metadata":{"papermill":{"duration":0.035922,"end_time":"2024-12-12T20:18:48.348136","exception":false,"start_time":"2024-12-12T20:18:48.312214","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission","metadata":{"execution":{"iopub.status.busy":"2024-12-19T19:00:59.944084Z","iopub.execute_input":"2024-12-19T19:00:59.944980Z","iopub.status.idle":"2024-12-19T19:00:59.955881Z","shell.execute_reply.started":"2024-12-19T19:00:59.944944Z","shell.execute_reply":"2024-12-19T19:00:59.954783Z"},"papermill":{"duration":0.047855,"end_time":"2024-12-12T20:18:48.432006","exception":false,"start_time":"2024-12-12T20:18:48.384151","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n# Hyperparameter Tuning\n\n- **LightGBM Parameters**: Hyperparameters such as `learning_rate`, `max_depth`, `num_leaves`, and `feature_fraction` are tuned to improve the performance of the LightGBM model. These parameters control the complexity of the model and its ability to generalize to new data.\n- **XGBoost and CatBoost Parameters**: Similar tuning is applied for XGBoost and CatBoost, adjusting parameters such as `n_estimators`, `max_depth`, `learning_rate`, `subsample`, and `regularization` terms (`reg_alpha`, `reg_lambda`). These help in controlling overfitting and ensuring the model's robustness.","metadata":{"papermill":{"duration":0.035531,"end_time":"2024-12-12T20:18:48.503391","exception":false,"start_time":"2024-12-12T20:18:48.46786","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Model parameters for LightGBM\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01,  # Increased from 2.68e-06\n    'device': 'cpu'\n\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED,\n    'tree_method': 'gpu_hist',\n\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10,  # Increase this value\n    'task_type': 'GPU'\n\n}","metadata":{"execution":{"iopub.status.busy":"2024-12-19T19:01:06.303190Z","iopub.execute_input":"2024-12-19T19:01:06.304129Z","iopub.status.idle":"2024-12-19T19:01:06.309714Z","shell.execute_reply.started":"2024-12-19T19:01:06.304093Z","shell.execute_reply":"2024-12-19T19:01:06.308654Z"},"papermill":{"duration":0.042924,"end_time":"2024-12-12T20:18:48.581996","exception":false,"start_time":"2024-12-12T20:18:48.539072","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# New: TabNet\n\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import train_test_split\nfrom pytorch_tabnet.callbacks import Callback\nimport os\nimport torch\nfrom pytorch_tabnet.callbacks import Callback\n\nclass TabNetWrapper(BaseEstimator, RegressorMixin):\n    def __init__(self, **kwargs):\n        self.model = TabNetRegressor(**kwargs)\n        self.kwargs = kwargs\n        self.imputer = SimpleImputer(strategy='median')\n        self.best_model_path = 'best_tabnet_model.pt'\n        \n    def fit(self, X, y):\n        # Handle missing values\n        X_imputed = self.imputer.fit_transform(X)\n        \n        if hasattr(y, 'values'):\n            y = y.values\n            \n        # Create internal validation set\n        X_train, X_valid, y_train, y_valid = train_test_split(\n            X_imputed, \n            y, \n            test_size=0.2,\n            random_state=42\n        )\n        \n        # Train TabNet model\n        history = self.model.fit(\n            X_train=X_train,\n            y_train=y_train.reshape(-1, 1),\n            eval_set=[(X_valid, y_valid.reshape(-1, 1))],\n            eval_name=['valid'],\n            eval_metric=['mse'],\n            max_epochs=200,\n            patience=20,\n            batch_size=1024,\n            virtual_batch_size=128,\n            num_workers=0,\n            drop_last=False,\n            callbacks=[\n                TabNetPretrainedModelCheckpoint(\n                    filepath=self.best_model_path,\n                    monitor='valid_mse',\n                    mode='min',\n                    save_best_only=True,\n                    verbose=True\n                )\n            ]\n        )\n        \n        # Load the best model\n        if os.path.exists(self.best_model_path):\n            self.model.load_model(self.best_model_path)\n            os.remove(self.best_model_path)  # Remove temporary file\n        \n        return self\n    \n    def predict(self, X):\n        X_imputed = self.imputer.transform(X)\n        return self.model.predict(X_imputed).flatten()\n    \n    def __deepcopy__(self, memo):\n        # Add deepcopy support for scikit-learn\n        cls = self.__class__\n        result = cls.__new__(cls)\n        memo[id(self)] = result\n        for k, v in self.__dict__.items():\n            setattr(result, k, deepcopy(v, memo))\n        return result\n\n# TabNet hyperparameters\nTabNet_Params = {\n    'n_d': 64,              # Width of the decision prediction layer\n    'n_a': 64,              # Width of the attention embedding for each step\n    'n_steps': 5,           # Number of steps in the architecture\n    'gamma': 1.5,           # Coefficient for feature selection regularization\n    'n_independent': 2,     # Number of independent GLU layer in each GLU block\n    'n_shared': 2,          # Number of shared GLU layer in each GLU block\n    'lambda_sparse': 1e-4,  # Sparsity regularization\n    'optimizer_fn': torch.optim.Adam,\n    'optimizer_params': dict(lr=2e-2, weight_decay=1e-5),\n    'mask_type': 'entmax',\n    'scheduler_params': dict(mode=\"min\", patience=10, min_lr=1e-5, factor=0.5),\n    'scheduler_fn': torch.optim.lr_scheduler.ReduceLROnPlateau,\n    'verbose': 1,\n    'device_name': 'cuda' if torch.cuda.is_available() else 'cpu'\n}\n\nclass TabNetPretrainedModelCheckpoint(Callback):\n    def __init__(self, filepath, monitor='val_loss', mode='min', \n                 save_best_only=True, verbose=1):\n        super().__init__()  # Initialize parent class\n        self.filepath = filepath\n        self.monitor = monitor\n        self.mode = mode\n        self.save_best_only = save_best_only\n        self.verbose = verbose\n        self.best = float('inf') if mode == 'min' else -float('inf')\n        \n    def on_train_begin(self, logs=None):\n        self.model = self.trainer  # Use trainer itself as model\n        \n    def on_epoch_end(self, epoch, logs=None):\n        logs = logs or {}\n        current = logs.get(self.monitor)\n        if current is None:\n            return\n        \n        # Check if current metric is better than best\n        if (self.mode == 'min' and current < self.best) or \\\n           (self.mode == 'max' and current > self.best):\n            if self.verbose:\n                print(f'\\nEpoch {epoch}: {self.monitor} improved from {self.best:.4f} to {current:.4f}')\n            self.best = current\n            if self.save_best_only:\n                self.model.save_model(self.filepath)  # Save the entire model","metadata":{"execution":{"iopub.status.busy":"2024-12-19T19:01:08.991444Z","iopub.execute_input":"2024-12-19T19:01:08.992253Z","iopub.status.idle":"2024-12-19T19:01:09.006596Z","shell.execute_reply.started":"2024-12-19T19:01:08.992216Z","shell.execute_reply":"2024-12-19T19:01:09.005613Z"},"papermill":{"duration":0.051218,"end_time":"2024-12-12T20:18:48.669191","exception":false,"start_time":"2024-12-12T20:18:48.617973","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Ensemble Learning and Submission Preparation\n\n- **Ensemble Learning**: The model uses a **Voting Regressor**, which combines the predictions from LightGBM, XGBoost, and CatBoost. This approach is beneficial as it leverages the strengths of multiple models, reducing overfitting and improving overall model performance.\n- **Out-of-Fold (OOF) Predictions**: During cross-validation, out-of-fold predictions are generated for the training set, which helps in model evaluation without data leakage.\n- **Kappa Optimizer**: The Kappa Optimizer ensures that the predicted values are as close to the actual values as possible by adjusting the thresholds used to convert raw model outputs into class labels.\n- **Test Set Predictions**: After the model is trained and thresholds are optimized, the test dataset is processed, and predictions are generated using the ensemble model. These predictions are converted into the appropriate format for submission.\n- **Submission File Creation**: The predictions are saved in a CSV file following the required format for submission (e.g., for a Kaggle competition), which includes columns like `id` and `sii` (Severity Impairment Index).","metadata":{"papermill":{"duration":0.035821,"end_time":"2024-12-12T20:18:48.741796","exception":false,"start_time":"2024-12-12T20:18:48.705975","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Final Results and Performance Metrics\n\n- **Train and Validation Scores**: After training across multiple folds, the mean Quadratic Weighted Kappa (QWK) score is calculated for both the training and validation datasets, providing an indicator of model performance. \n- **Optimized QWK Score**: The final optimized QWK score after threshold tuning is displayed, showcasing the model's ability to predict the severity levels effectively.\n- **Test Predictions**: The test set predictions are evaluated, and a breakdown of the predicted severity levels (None, Mild, Moderate, Severe) is shown, along with their respective counts.","metadata":{"papermill":{"duration":0.038161,"end_time":"2024-12-12T20:18:48.817502","exception":false,"start_time":"2024-12-12T20:18:48.779341","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\nTabNet_Model = TabNetWrapper(**TabNet_Params) # New","metadata":{"execution":{"iopub.status.busy":"2024-12-19T19:01:13.665740Z","iopub.execute_input":"2024-12-19T19:01:13.666562Z","iopub.status.idle":"2024-12-19T19:01:13.678284Z","shell.execute_reply.started":"2024-12-19T19:01:13.666517Z","shell.execute_reply":"2024-12-19T19:01:13.677337Z"},"papermill":{"duration":0.050985,"end_time":"2024-12-12T20:18:48.906901","exception":false,"start_time":"2024-12-12T20:18:48.855916","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n# **》》》Model1.Train**\n---","metadata":{"papermill":{"duration":0.03675,"end_time":"2024-12-12T20:18:48.979839","exception":false,"start_time":"2024-12-12T20:18:48.943089","status":"completed"},"tags":[]}},{"cell_type":"code","source":"voting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    ('tabnet', TabNet_Model)\n],weights=[4.0,4.0,5.0,4.0])\n\nSubmission1 = TrainML(voting_model, test)\n\nSubmission1","metadata":{"execution":{"iopub.status.busy":"2024-12-19T19:01:16.058424Z","iopub.execute_input":"2024-12-19T19:01:16.059015Z","iopub.status.idle":"2024-12-19T19:02:26.632424Z","shell.execute_reply.started":"2024-12-19T19:01:16.058978Z","shell.execute_reply":"2024-12-19T19:02:26.631633Z"},"papermill":{"duration":74.74773,"end_time":"2024-12-12T20:20:03.763394","exception":false,"start_time":"2024-12-12T20:18:49.015664","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"```\n],weights=[5.0,4.0,4.0,4.0])\nMean Train QWK --> 0.7424\nMean Validation QWK ---> 0.4735\n----> || Optimized QWK SCORE ::  0.533\n\n```","metadata":{"papermill":{"duration":0.037036,"end_time":"2024-12-12T20:20:03.838467","exception":false,"start_time":"2024-12-12T20:20:03.801431","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"---\n# **》》》Model2**\n---","metadata":{"papermill":{"duration":0.036329,"end_time":"2024-12-12T20:20:03.911459","exception":false,"start_time":"2024-12-12T20:20:03.87513","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    df = df.loc[:, ['X', 'Y', 'Z', 'enmo', 'anglez']]\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n        \ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntrain_ts = train_ts.merge(train_ts_7, on='id', how='left')\ntest_ts = test_ts.merge(test_ts_7, on='id', how='left')\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.49, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    thresholds = KappaOPtimizer.x\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, thresholds)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    fold_weights = [1.25, 1.0, 1.0, 1.0, 1.0]\n    tpm = test_preds.dot(fold_weights) / np.sum(fold_weights)\n    tpTuned = threshold_Rounder(tpm, thresholds)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission\n\n# Model parameters for LightGBM\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01  # Increased from 2.68e-06\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'cat_features': cat_c,\n    'verbose': 0,\n    'l2_leaf_reg': 10  # Increase this value\n}\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model)\n])\n\n# Train the ensemble model\nSubmission2 = TrainML(voting_model, test)\n\n# Save submission\n#Submission2.to_csv('submission.csv', index=False)\nSubmission2","metadata":{"execution":{"iopub.status.busy":"2024-12-19T19:04:36.834720Z","iopub.execute_input":"2024-12-19T19:04:36.835081Z","iopub.status.idle":"2024-12-19T19:06:55.102959Z","shell.execute_reply.started":"2024-12-19T19:04:36.835051Z","shell.execute_reply":"2024-12-19T19:06:55.101975Z"},"papermill":{"duration":125.974049,"end_time":"2024-12-12T20:22:09.922074","exception":false,"start_time":"2024-12-12T20:20:03.948025","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n# **》》》Model3**\n---","metadata":{"papermill":{"duration":0.038492,"end_time":"2024-12-12T20:22:10.005701","exception":false,"start_time":"2024-12-12T20:22:09.967209","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntrain_ts = train_ts.merge(train_ts_7, on='id', how='left')\ntest_ts = test_ts.merge(test_ts_7, on='id', how='left')\n\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    thresholds = KappaOPtimizer.x\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, thresholds)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tp_rounded = threshold_Rounder(tpm, thresholds)\n\n    return tp_rounded\n\nimputer = SimpleImputer(strategy='median')\n\nensemble = VotingRegressor(estimators=[\n    ('lgb', Pipeline(steps=[('imputer', imputer), ('regressor', LGBMRegressor(random_state=SEED))])),\n    ('xgb', Pipeline(steps=[('imputer', imputer), ('regressor', XGBRegressor(random_state=SEED))])),\n    ('cat', Pipeline(steps=[('imputer', imputer), ('regressor', CatBoostRegressor(random_state=SEED, silent=True))])),\n    ('rf', Pipeline(steps=[('imputer', imputer), ('regressor', RandomForestRegressor(random_state=SEED))])),\n    ('gb', Pipeline(steps=[('imputer', imputer), ('regressor', GradientBoostingRegressor(random_state=SEED))]))\n])\n\nSubmission3 = TrainML(ensemble, test)\nSubmission3 = pd.DataFrame({\n    'id': sample['id'],\n    'sii': Submission3\n})\n\nSubmission3","metadata":{"execution":{"iopub.status.busy":"2024-12-19T19:12:35.409594Z","iopub.execute_input":"2024-12-19T19:12:35.410018Z","iopub.status.idle":"2024-12-19T19:15:10.477305Z","shell.execute_reply.started":"2024-12-19T19:12:35.409984Z","shell.execute_reply":"2024-12-19T19:15:10.475980Z"},"papermill":{"duration":195.852934,"end_time":"2024-12-12T20:25:25.901091","exception":false,"start_time":"2024-12-12T20:22:10.048157","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Final Ensemble","metadata":{"papermill":{"duration":0.037551,"end_time":"2024-12-12T20:25:25.97673","exception":false,"start_time":"2024-12-12T20:25:25.939179","status":"completed"},"tags":[]}},{"cell_type":"code","source":"sub1 = Submission1\nsub2 = Submission2\nsub3 = Submission3\n\nsub1 = sub1.sort_values(by='id').reset_index(drop=True)\nsub2 = sub2.sort_values(by='id').reset_index(drop=True)\nsub3 = sub3.sort_values(by='id').reset_index(drop=True)\n\ncombined = pd.DataFrame({\n    'id': sub1['id'],\n    'sii_1': sub1['sii'],\n    'sii_2': sub2['sii'],\n    'sii_3': sub3['sii']\n})\n\ndef majority_vote(row):\n    return row.mode()[0]\n\ncombined['final_sii'] = combined[['sii_1', 'sii_2', 'sii_3']].apply(majority_vote, axis=1)\n\nfinal_submission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\n\nfinal_submission.to_csv('submission.csv', index=False)\n\nprint(\"Majority voting completed and saved to 'Final_Submission.csv'\")","metadata":{"execution":{"iopub.status.busy":"2024-12-19T19:10:59.280887Z","iopub.execute_input":"2024-12-19T19:10:59.283520Z","iopub.status.idle":"2024-12-19T19:10:59.317266Z","shell.execute_reply.started":"2024-12-19T19:10:59.283467Z","shell.execute_reply":"2024-12-19T19:10:59.316102Z"},"papermill":{"duration":0.064111,"end_time":"2024-12-12T20:25:26.078163","exception":false,"start_time":"2024-12-12T20:25:26.014052","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_submission","metadata":{"execution":{"iopub.status.busy":"2024-12-19T19:11:03.329180Z","iopub.execute_input":"2024-12-19T19:11:03.329963Z","iopub.status.idle":"2024-12-19T19:11:03.338512Z","shell.execute_reply.started":"2024-12-19T19:11:03.329933Z","shell.execute_reply":"2024-12-19T19:11:03.337577Z"},"papermill":{"duration":0.049067,"end_time":"2024-12-12T20:25:26.165082","exception":false,"start_time":"2024-12-12T20:25:26.116015","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub1 = Submission1\nsub2 = Submission2\nsub3 = Submission3\n\nsub1 = sub1.sort_values(by='id').reset_index(drop=True)\nsub2 = sub2.sort_values(by='id').reset_index(drop=True)\nsub3 = sub3.sort_values(by='id').reset_index(drop=True)\nfinal_submission = final_submission.sort_values(by='id').reset_index(drop=True)\ncombined = pd.DataFrame({\n    'id': sub1['id'],\n    'sii_1': sub1['sii'],\n    'sii_2': sub2['sii'],\n    'sii_3': sub3['sii'],\n    'final': final_submission['sii']\n})\n\ncombined","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T19:16:42.485863Z","iopub.execute_input":"2024-12-19T19:16:42.486434Z","iopub.status.idle":"2024-12-19T19:16:42.501670Z","shell.execute_reply.started":"2024-12-19T19:16:42.486401Z","shell.execute_reply":"2024-12-19T19:16:42.500727Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}