{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 12S22045 - Cintya Sitorus","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-07T13:38:55.029117Z","iopub.execute_input":"2024-12-07T13:38:55.029741Z","iopub.status.idle":"2024-12-07T13:38:56.157240Z","shell.execute_reply.started":"2024-12-07T13:38:55.029689Z","shell.execute_reply":"2024-12-07T13:38:56.155914Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom colorama import Fore, Style\n\nfrom lightgbm import LGBMRegressor\nfrom catboost import CatBoostRegressor\nfrom xgboost import XGBRegressor\n\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.base import clone\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.impute import KNNImputer\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nfrom scipy.optimize import minimize\nfrom sklearn.ensemble import VotingRegressor\n\nimport torch\nimport os\nimport torch.nn as nn\nfrom concurrent.futures import ThreadPoolExecutor\nimport torch.optim as optim\nfrom keras.optimizers import Adam\n\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T13:38:56.162548Z","iopub.execute_input":"2024-12-07T13:38:56.162980Z","iopub.status.idle":"2024-12-07T13:38:56.170954Z","shell.execute_reply.started":"2024-12-07T13:38:56.162946Z","shell.execute_reply":"2024-12-07T13:38:56.169591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Load Data\ndf = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\ndf = df.dropna(subset=['sii']) # keeping labeled values only\ntest = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")\n\n\n\nseason_mapping = {\n    'Winter': -1,\n    'Spring': -0.5,\n    'Summer': 0.5,\n    'Fall': 1\n}\n\n# mapping non-string values\ndf = df.replace(season_mapping)\ntest = test.replace(season_mapping)\n\n# # dropping questions not in test dataset\ntest_missing_columns = set(df.columns) - set(test.columns)\nfor col in test_missing_columns:\n    if col != 'sii':  # Retain the target column for training\n        df.drop(columns=col, inplace=True)\n# for later use\ntrain_ids = df['id']\ntest_ids = test['id']\ntrain_labels = df['sii']\ndf = df.drop(columns=['id'])\n\n# # Drop weakly correlated features\ndf = df.drop(columns=['sii'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T13:38:56.172589Z","iopub.execute_input":"2024-12-07T13:38:56.173028Z","iopub.status.idle":"2024-12-07T13:38:56.286038Z","shell.execute_reply.started":"2024-12-07T13:38:56.172990Z","shell.execute_reply":"2024-12-07T13:38:56.284719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#KKN missing data imputation\nimputer = KNNImputer(n_neighbors=4)  # k=4\nimputed_data = imputer.fit_transform(df)\n\ntrain = pd.DataFrame(imputed_data, columns=df.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T13:38:56.289609Z","iopub.execute_input":"2024-12-07T13:38:56.290227Z","iopub.status.idle":"2024-12-07T13:38:59.463566Z","shell.execute_reply.started":"2024-12-07T13:38:56.290167Z","shell.execute_reply":"2024-12-07T13:38:59.462326Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Parquet file opening functions\ndef load_time_series(dirname) -> pd.DataFrame:\n    # opening parquet files and returning dataframe.\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T13:38:59.465348Z","iopub.execute_input":"2024-12-07T13:38:59.465737Z","iopub.status.idle":"2024-12-07T13:38:59.473747Z","shell.execute_reply.started":"2024-12-07T13:38:59.465700Z","shell.execute_reply":"2024-12-07T13:38:59.472428Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Autoeconder for accelerometer data- Compressing accelerometer tabular data to N features and merging them on orginal dataset\nclass AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim*3),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*3, encoding_dim*2),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*2, encoding_dim),\n            nn.ReLU()\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim*2),\n            nn.ReLU(),\n            nn.Linear(input_dim*2, input_dim*3),\n            nn.ReLU(),\n            nn.Linear(input_dim*3, input_dim),\n            nn.Sigmoid()\n        )\n        \n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded\n    \ndef perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    \n    data_tensor = torch.FloatTensor(df_scaled)\n    \n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n    \n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(autoencoder.parameters())\n    \n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n            \n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n                 \n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n        \n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n    \n    return df_encoded","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T13:38:59.537092Z","iopub.execute_input":"2024-12-07T13:38:59.537541Z","iopub.status.idle":"2024-12-07T13:38:59.550675Z","shell.execute_reply.started":"2024-12-07T13:38:59.537492Z","shell.execute_reply":"2024-12-07T13:38:59.549520Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# qwk, threshold rounder and prediction evaluator\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n# threshold rounder\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n# prediction evaluation using qwk function\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T13:38:59.552360Z","iopub.execute_input":"2024-12-07T13:38:59.552842Z","iopub.status.idle":"2024-12-07T13:38:59.568793Z","shell.execute_reply.started":"2024-12-07T13:38:59.552793Z","shell.execute_reply":"2024-12-07T13:38:59.567653Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#test = test.drop(columns='id')\n## accelerometer data\nprint('Loading train timeseries data...')\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\nprint('Loading test timeseries data...')\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\nprint(f'Shape of Train accelerometer data: {train_ts.shape}')\nprint(f'Shape of Test accelerometer data: {test_ts.shape}')\n\ndf_train_ts = train_ts.drop('id', axis=1)\ndf_test_ts = test_ts.drop('id', axis=1)\n\ntrain_ts_encoded = perform_autoencoder(df_train_ts, encoding_dim=60, epochs=100, batch_size=32)\ntest_ts_encoded = perform_autoencoder(df_test_ts, encoding_dim=60, epochs=100, batch_size=32)\n\ntime_series_cols = train_ts_encoded.columns.tolist()\ntrain_ts_encoded[\"id\"]=train_ts[\"id\"]\ntest_ts_encoded['id']=test_ts[\"id\"]\n\n# re-adding primary keys\ntrain['id'] = train_ids\ntest['id'] = test_ids\ntrain = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')\n\ntrain['id']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T13:38:59.572809Z","iopub.execute_input":"2024-12-07T13:38:59.573359Z","iopub.status.idle":"2024-12-07T13:41:39.774681Z","shell.execute_reply.started":"2024-12-07T13:38:59.573304Z","shell.execute_reply":"2024-12-07T13:41:39.773483Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# funciton that trains any regressor model using kfold cross validation, k hard coded = 5\ndef TrainML(model_class, test_data) -> list[int]:\n    global train\n    \n\n    \n    #train = train.drop(columns=['id'])\n    test_data = test_data.drop(columns=['id'])\n    train = train[test_data.columns]\n\n    \n\n    \n    X = train # .drop(['sii'], axis=1)\n    y = train_labels\n\n    imputer = SimpleImputer(strategy='mean')\n    X = pd.DataFrame(imputer.fit_transform(train), columns=train.columns, index=train.index)\n\n    # Ensure test_data also remains a DataFrame\n    test_data = pd.DataFrame(imputer.transform(test_data), columns=test_data.columns, index=test_data.index)\n\n    n_splits=5\n    random_state=42\n    \n    ################    \n    scaler = StandardScaler()\n    \n    scaler.fit(X)\n\n\n    X = pd.DataFrame(scaler.transform(X), columns=X.columns)\n   \n    #ids are stored in test_ids variable\n    test_data = pd.DataFrame(scaler.transform(test_data), columns=test_data.columns)\n    #############\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=random_state)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        \n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tp_rounded = threshold_Rounder(tpm, KappaOPtimizer.x)\n    return tp_rounded.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T13:41:39.776083Z","iopub.execute_input":"2024-12-07T13:41:39.776445Z","iopub.status.idle":"2024-12-07T13:41:39.792962Z","shell.execute_reply.started":"2024-12-07T13:41:39.776412Z","shell.execute_reply":"2024-12-07T13:41:39.791037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#parameters\nLGBM_params = {\n    'n_estimators': 300,\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  \n    'lambda_l2': 0.01\n}\n\n\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 300,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  \n    'reg_lambda': 5,  \n    'random_state': 42,\n    'tree_method': 'exact'\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 300,\n    'random_seed': 42,\n    'verbose': 0,\n    'l2_leaf_reg': 10  \n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T13:41:39.794876Z","iopub.execute_input":"2024-12-07T13:41:39.795412Z","iopub.status.idle":"2024-12-07T13:41:39.814018Z","shell.execute_reply.started":"2024-12-07T13:41:39.795356Z","shell.execute_reply":"2024-12-07T13:41:39.812807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Model Declaration\nLight = LGBMRegressor(**LGBM_params, random_state=42, verbose=-1)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n],\n     weights=[0.6, 0.4, 0.6] # 0.3 0.5 0.2\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T13:41:39.815585Z","iopub.execute_input":"2024-12-07T13:41:39.817238Z","iopub.status.idle":"2024-12-07T13:41:39.829087Z","shell.execute_reply.started":"2024-12-07T13:41:39.817178Z","shell.execute_reply":"2024-12-07T13:41:39.827912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# XGBOOST\nxgb_preds = TrainML(model_class=XGB_Model, test_data=test)\nsub = pd.DataFrame({\n    \n    'id'   : test_ids,\n    \n    'sii': xgb_preds\n})\nxgb_preds\nsub","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T13:41:39.831085Z","iopub.execute_input":"2024-12-07T13:41:39.831666Z","iopub.status.idle":"2024-12-07T13:42:18.848650Z","shell.execute_reply.started":"2024-12-07T13:41:39.831611Z","shell.execute_reply":"2024-12-07T13:42:18.847464Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Light Gradient Boosting Machine Model\nlgbm_preds = TrainML(model_class=Light, test_data=test)\nsub = pd.DataFrame({\n    \n    'id'   : test_ids,\n    \n    'sii': lgbm_preds\n})\n\nsub","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T13:42:18.850266Z","iopub.execute_input":"2024-12-07T13:42:18.850835Z","iopub.status.idle":"2024-12-07T13:42:28.539251Z","shell.execute_reply.started":"2024-12-07T13:42:18.850783Z","shell.execute_reply":"2024-12-07T13:42:28.538019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Catboost Model\ncat_preds = TrainML(model_class=CatBoost_Model, test_data=test)\nsub = pd.DataFrame({\n    \n    'id'   : test_ids,\n    \n    'sii': cat_preds\n})\n\nsub","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T13:42:28.540826Z","iopub.execute_input":"2024-12-07T13:42:28.541341Z","iopub.status.idle":"2024-12-07T13:42:48.448977Z","shell.execute_reply.started":"2024-12-07T13:42:28.541287Z","shell.execute_reply":"2024-12-07T13:42:48.447768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"vote_preds = TrainML(model_class=voting_model, test_data=test)\nfinal_sub = pd.DataFrame({\n    \n    'id'   : test_ids,\n    'sii': vote_preds\n})\n\n\nfinal_sub.to_csv('submission.csv', index=False)\nfinal_sub","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T13:42:48.450460Z","iopub.execute_input":"2024-12-07T13:42:48.450905Z","iopub.status.idle":"2024-12-07T13:43:57.727817Z","shell.execute_reply.started":"2024-12-07T13:42:48.450858Z","shell.execute_reply":"2024-12-07T13:43:57.726452Z"}},"outputs":[],"execution_count":null}]}