{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-24T07:56:31.508911Z","iopub.execute_input":"2024-12-24T07:56:31.509206Z","iopub.status.idle":"2024-12-24T07:56:36.29997Z","shell.execute_reply.started":"2024-12-24T07:56:31.509182Z","shell.execute_reply":"2024-12-24T07:56:36.299067Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step1: Import libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport re\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T07:57:12.02535Z","iopub.execute_input":"2024-12-24T07:57:12.025646Z","iopub.status.idle":"2024-12-24T07:57:12.02996Z","shell.execute_reply.started":"2024-12-24T07:57:12.025628Z","shell.execute_reply":"2024-12-24T07:57:12.028366Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport polars as pl\nfrom IPython.display import clear_output\nfrom tqdm import tqdm\nfrom concurrent.futures import ThreadPoolExecutor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T07:57:28.573933Z","iopub.execute_input":"2024-12-24T07:57:28.574322Z","iopub.status.idle":"2024-12-24T07:57:29.002243Z","shell.execute_reply.started":"2024-12-24T07:57:28.574294Z","shell.execute_reply":"2024-12-24T07:57:29.000918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import lightgbm as lgb\nfrom catboost import CatBoostRegressor, CatBoostClassifier\nfrom xgboost import XGBRegressor\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.base import clone\nfrom sklearn.model_selection import *\nfrom sklearn.metrics import *\nfrom scipy.optimize import minimize\nfrom scipy.stats import mode","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T07:57:34.600641Z","iopub.execute_input":"2024-12-24T07:57:34.600997Z","iopub.status.idle":"2024-12-24T07:57:38.827496Z","shell.execute_reply.started":"2024-12-24T07:57:34.600971Z","shell.execute_reply":"2024-12-24T07:57:38.826501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import optuna\nfrom colorama import Fore, Style\npd.options.display.max_columns = None\nSEED = 42\nn_splits = 5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T07:57:54.402816Z","iopub.execute_input":"2024-12-24T07:57:54.403435Z","iopub.status.idle":"2024-12-24T07:57:54.683563Z","shell.execute_reply.started":"2024-12-24T07:57:54.403401Z","shell.execute_reply":"2024-12-24T07:57:54.682467Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step2:Import dataset","metadata":{}},{"cell_type":"code","source":"# Load Parquet and Process Files\ndef process_file(filename, dirname):\n    file_path = os.path.join(dirname, filename, 'part-0.parquet')\n    df = pd.read_parquet(file_path)\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(\n            executor.map(lambda fname: process_file(fname, dirname), ids),\n            total=len(ids)\n        ))\n    \n    stats, indexes = zip(*results)\n    df = pd.DataFrame(stats, columns=[f\"Stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n\n# Load Data\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\n# Merge and Prepare Data\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain.drop('id', axis=1, inplace=True)\ntest.drop('id', axis=1, inplace=True)\n\n# Features\nfeaturesCols = [\n    'Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-Season',\n    'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI', 'Physical-Height',\n    'Physical-Weight', 'Physical-Waist_Circumference', 'Physical-Diastolic_BP',\n    'Physical-HeartRate', 'Physical-Systolic_BP', 'Fitness_Endurance-Season',\n    'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins',\n    'Fitness_Endurance-Time_Sec', 'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone',\n    'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone',\n    'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone',\n    'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n    'BIA-Season', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n    'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI',\n    'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num', 'BIA-BIA_ICW',\n    'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PAQ_A-Season',\n    'PAQ_A-PAQ_A_Total', 'PAQ_C-Season', 'PAQ_C-PAQ_C_Total', 'SDS-Season',\n    'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n    'PreInt_EduHx-computerinternet_hoursday', 'sii'\n]\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain.dropna(subset=['sii'], inplace=True)\n\n# Categorical Columns\ncat_c = [\n    'Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season',\n    'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', 'PAQ_A-Season',\n    'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season'\n]\n\n# Update and Encode Categorical Columns\ndef update_and_encode(df):\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing').astype('category')\n        mapping = {value: idx for idx, value in enumerate(df[c].cat.categories)}\n        df[c] = df[c].replace(mapping).astype(int)\n    return df\n\ntrain = update_and_encode(train)\ntest = update_and_encode(test)\n\nprint(f'Train Shape : {train.shape} || Test Shape : {test.shape}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T07:59:08.942043Z","iopub.execute_input":"2024-12-24T07:59:08.942333Z","iopub.status.idle":"2024-12-24T08:00:22.244131Z","shell.execute_reply.started":"2024-12-24T07:59:08.942313Z","shell.execute_reply":"2024-12-24T08:00:22.242732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T08:00:22.245713Z","iopub.execute_input":"2024-12-24T08:00:22.246127Z","iopub.status.idle":"2024-12-24T08:00:22.331788Z","shell.execute_reply.started":"2024-12-24T08:00:22.246099Z","shell.execute_reply":"2024-12-24T08:00:22.330599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.tail()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T08:00:22.33325Z","iopub.execute_input":"2024-12-24T08:00:22.333564Z","iopub.status.idle":"2024-12-24T08:00:22.416057Z","shell.execute_reply.started":"2024-12-24T08:00:22.333535Z","shell.execute_reply":"2024-12-24T08:00:22.414595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T08:00:22.417427Z","iopub.execute_input":"2024-12-24T08:00:22.417758Z","iopub.status.idle":"2024-12-24T08:00:22.437104Z","shell.execute_reply.started":"2024-12-24T08:00:22.417731Z","shell.execute_reply":"2024-12-24T08:00:22.435625Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T08:00:36.151838Z","iopub.execute_input":"2024-12-24T08:00:36.153807Z","iopub.status.idle":"2024-12-24T08:00:36.514968Z","shell.execute_reply.started":"2024-12-24T08:00:36.153752Z","shell.execute_reply":"2024-12-24T08:00:36.513367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\nX = train.drop(['sii'], axis=1)\ny = train['sii']\n\ndef TrainML(model_class, test_data, seed_list):\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=False,)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        \n        random_seed = np.random.choice(seed_list)\n        \n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        if hasattr(model, 'random_state'):\n            model.set_params(random_state=random_seed)\n\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Train : {np.mean(train_S):.4f}\")\n    print(f\"Validation : {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead') \n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized : {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T08:06:52.357546Z","iopub.execute_input":"2024-12-24T08:06:52.357857Z","iopub.status.idle":"2024-12-24T08:06:52.369621Z","shell.execute_reply.started":"2024-12-24T08:06:52.357838Z","shell.execute_reply":"2024-12-24T08:06:52.368792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import lightgbm as lgb\nimport numpy as np\n\n# Define the latest parameters for the model\nLatestParams = {\n    'learning_rate': 0.03755757104848504,\n    'max_depth': 12,\n    'num_leaves': 18,\n    'min_data_in_leaf': 3,\n    'feature_fraction': 0.723690362968002,\n    'bagging_fraction': 0.688232590484764,\n    'bagging_freq': 5,\n    'lambda_l1': 0.18512987285245963,\n    'lambda_l2': 0.18435628737334625\n}\n\n# Define the random seed list\nseed_list = [42, 0, 2024, 7269173, 1234]\n\n# Initialize the LightGBM Regressor with the latest parameters\nLight = lgb.LGBMRegressor(**LatestParams, verbose=-1, n_estimators=200)\n\n# Function to train and predict using a given model and dataset\nSubmissionEstimator1 = TrainML(Light, test, seed_list)\n\n# Print the submission prediction (optional)\nprint(SubmissionEstimator1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T08:06:54.466076Z","iopub.execute_input":"2024-12-24T08:06:54.466468Z","iopub.status.idle":"2024-12-24T08:07:00.661357Z","shell.execute_reply.started":"2024-12-24T08:06:54.466439Z","shell.execute_reply":"2024-12-24T08:07:00.659916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(SubmissionEstimator1['sii'].value_counts())\nSubmissionEstimator1.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T08:09:02.107567Z","iopub.execute_input":"2024-12-24T08:09:02.107951Z","iopub.status.idle":"2024-12-24T08:09:02.120217Z","shell.execute_reply.started":"2024-12-24T08:09:02.107929Z","shell.execute_reply":"2024-12-24T08:09:02.118778Z"}},"outputs":[],"execution_count":null}]}