{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom colorama import Fore, Style\n\nfrom lightgbm import LGBMRegressor\nfrom catboost import CatBoostRegressor\nfrom xgboost import XGBRegressor\n\nfrom sklearn.base import clone\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.impute import KNNImputer\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nfrom scipy.optimize import minimize\nfrom sklearn.ensemble import VotingRegressor\nfrom concurrent.futures import ThreadPoolExecutor\nimport os\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import make_scorer\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom tqdm import tqdm\nfrom sklearn.ensemble import RandomForestRegressor\n\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, BatchNormalization, Conv1D, Flatten, MaxPooling1D, Embedding, GlobalMaxPooling1D, Input\nfrom tensorflow.keras.models import Model","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-20T19:12:24.164426Z","iopub.execute_input":"2024-12-20T19:12:24.164874Z","execution_failed":"2024-12-20T19:12:26.817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# qwk score\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n\n# threshold rounder\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\n\n# prediction evaluation using qwk function\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    # opening parquet files and returning dataframe.\n    ids = os.listdir(dirname)\n\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n\n    stats, indexes = zip(*results)\n\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop(columns=['step', 'battery_voltage', 'non-wear_flag'], axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:49:27.685541Z","iopub.execute_input":"2024-12-19T09:49:27.686059Z","iopub.status.idle":"2024-12-19T09:49:27.715364Z","shell.execute_reply.started":"2024-12-19T09:49:27.686006Z","shell.execute_reply":"2024-12-19T09:49:27.714067Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim*3),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*3, encoding_dim*2),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*2, encoding_dim),\n            nn.ReLU()\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim*2),\n            nn.ReLU(),\n            nn.Linear(input_dim*2, input_dim*3),\n            nn.ReLU(),\n            nn.Linear(input_dim*3, input_dim),\n            nn.Sigmoid()\n        )\n        \n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded\n    \ndef perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    \n    data_tensor = torch.FloatTensor(df_scaled)\n    \n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n    \n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(autoencoder.parameters())\n    \n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n            \n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n                 \n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n        \n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n    \n    return df_encoded\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:49:27.716938Z","iopub.execute_input":"2024-12-19T09:49:27.717466Z","iopub.status.idle":"2024-12-19T09:49:27.744204Z","shell.execute_reply.started":"2024-12-19T09:49:27.71741Z","shell.execute_reply":"2024-12-19T09:49:27.742563Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# funciton that trains any regressor model using kfold cross validation, k hard coded = 5\ndef TrainML(model_class, test_data, acc=False) -> list[int]:\n    if not acc:\n        global train\n        X = train.drop(['sii'], axis=1)\n        y = train_labels\n        test_data = test_data[X.columns]  # Reorder test_data columns to match X\n\n        ################\n        scaler = StandardScaler()\n        scaler.fit(X)\n        X = pd.DataFrame(scaler.transform(X), columns=X.columns)\n\n        # ids are stored in test_ids variable\n        #test_data = test_data.drop(columns='id')\n        test_data = pd.DataFrame(scaler.transform(test_data), columns=test_data.columns)\n        #############\n    else:\n        global train_ts\n        y = train_ts['sii']\n        train_ts = train_ts.drop(columns=['id', 'sii'], axis=1)\n        X = train_ts\n    n_splits = 5\n    random_state = 42\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=random_state)\n\n    train_S = []\n    test_S = []\n\n    oof_non_rounded = np.zeros(len(y), dtype=float)\n    oof_rounded = np.zeros(len(y), dtype=int)\n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n\n        test_preds[:, fold] = model.predict(test_data)\n\n        print(f\"Fold {fold + 1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded),\n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n\n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tp_rounded = threshold_Rounder(tpm, KappaOPtimizer.x)\n    return tp_rounded.tolist(), tKappa","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:49:27.746969Z","iopub.execute_input":"2024-12-19T09:49:27.747376Z","iopub.status.idle":"2024-12-19T09:49:27.774513Z","shell.execute_reply.started":"2024-12-19T09:49:27.747338Z","shell.execute_reply":"2024-12-19T09:49:27.773167Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"LGBM_params = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,\n    'lambda_l2': 0.01\n}\n\n\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 300,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,\n    'reg_lambda': 5,\n    'random_state': 42,\n    'tree_method': 'exact'\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 400,\n    'random_seed': 42,\n    'verbose': 0,\n    'l2_leaf_reg': 10\n}\n\n#\nLGBM_params_acc = {\n\n    'learning_rate': 0.01,\n    'max_depth': 3,\n    'num_leaves': 31,\n    'min_child_samples': 10,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,\n    'lambda_l2': 0.01\n}\n\n\nXGB_Params_acc = {\n    'learning_rate': 0.01,\n    'max_depth': 3,\n    'n_estimators': 300,\n    'num_leaves': 31,\n    'min_child_samples': 10,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,\n    'reg_lambda': 5,\n    'random_state': 42,\n    'tree_method': 'exact'\n}\n\n\nCatBoost_Params_acc = {\n    'learning_rate': 0.01,\n    'depth': 3,\n    'min_child_samples': 10,\n    'num_leaves': 31,\n    'iterations': 300,\n    'random_seed': 42,\n    'verbose': 0,\n    'l2_leaf_reg': 10\n}","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:49:27.910921Z","iopub.execute_input":"2024-12-19T09:49:27.911362Z","iopub.status.idle":"2024-12-19T09:49:27.922702Z","shell.execute_reply.started":"2024-12-19T09:49:27.911324Z","shell.execute_reply":"2024-12-19T09:49:27.921451Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#####################################\ndf = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\ndf = df.dropna(subset=['sii']) # keeping labeled values only\ntest = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")\nseason_mapping = {\n    'Winter': -1,\n    'Spring': -0.5,\n    'Summer': 0.5,\n    'Fall': 1\n}\n# mapping non-string values\ndf = df.replace(season_mapping)\ntest = test.replace(season_mapping)\n\n# dropping questions not in test dataset\ntest_missing_columns = set(df.columns) - set(test.columns)\nfor col in test_missing_columns:\n    if col != 'sii':  # Retain the target column for training\n        df.drop(columns=col, inplace=True)\n# for later use\ntrain_ids = df['id']\ntest_ids = test['id']\ntrain_labels = df['sii']\n\n##################\ndf","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:49:27.92474Z","iopub.execute_input":"2024-12-19T09:49:27.925141Z","iopub.status.idle":"2024-12-19T09:49:28.067743Z","shell.execute_reply.started":"2024-12-19T09:49:27.925104Z","shell.execute_reply":"2024-12-19T09:49:28.066541Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"    print('Loading train timeseries data...')\n    train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\n    print('Loading test timeseries data...')\n    test_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n    print(f'Shape of Train accelerometer data: {train_ts.shape}')\n    print(f'Shape of Test accelerometer data: {test_ts.shape}')\n    \n    train_ts = pd.merge(train_ts, df[['id', 'sii']], how='left', on='id')\n    \n    ######################\n    Light = LGBMRegressor(**LGBM_params_acc, random_state=42, verbose=-1, n_estimators=300)\n    XGB_Model = XGBRegressor(**XGB_Params_acc)\n    CatBoost_Model = CatBoostRegressor(**CatBoost_Params_acc)\n    \n    voting_model = VotingRegressor(estimators=[\n        ('lightgbm', Light),\n        ('xgboost', XGB_Model),\n        ('catboost', CatBoost_Model)],\n         weights=[1, 1, 1]\n    )\n    ttsid = train_ts['id']\n    testtsid = test_ts['id']\n    \n    \n    train_ts = train_ts.drop(columns=['id'])\n    test_ts = test_ts.drop(columns=['id'])\n    \n    train_ts_encoded = perform_autoencoder(train_ts, encoding_dim=40, epochs=100, batch_size=32)\n    test_ts_encoded = perform_autoencoder(test_ts, encoding_dim=40, epochs=100, batch_size=32)\n    \n    train_ts_encoded['id'] = ttsid\n    test_ts_encoded['id'] = testtsid \n    \n    train = pd.merge(df, train_ts_encoded, how=\"left\", on='id')\n    test = pd.merge(test, test_ts_encoded, how=\"left\", on='id')\n    \n    # test_ts_ds = test_ts['id']\n    # test_ts = test_ts.drop(columns=['id'])\n    \n    # vote_preds = TrainML(model_class=voting_model, test_data=test_ts, acc=True)\n    # acc_sub = pd.DataFrame({\n    \n    #     'id': test_ts_ds,\n    #     'sii': vote_preds\n    # })\n    \n    # acc_sub\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:49:28.069052Z","iopub.execute_input":"2024-12-19T09:49:28.069477Z","iopub.status.idle":"2024-12-19T09:51:36.192423Z","shell.execute_reply.started":"2024-12-19T09:49:28.069441Z","shell.execute_reply":"2024-12-19T09:51:36.191171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Define parameter grid for LightGBM\n  \n# y = train_ts['sii']\n# train_ts = train_ts.drop(columns=['id', 'sii'], axis=1)\n# X = train_ts\n\n# param_grid = {\n#     'num_leaves': [31, 50, 70],\n#     'max_depth': [3, 5, 7],\n#     'learning_rate': [0.01, 0.05, 0.1],\n#     'min_child_samples': [10, 20, 50]\n# }\n# qwk_scorer = make_scorer(quadratic_weighted_kappa, greater_is_better=True)\n\n# # GridSearchCV for LightGBM\n# grid_search = GridSearchCV(\n#     estimator=CatBoostRegressor(random_state=42, n_estimators=100),\n#     param_grid=param_grid,\n#     scoring=qwk_scorer,  # Use QWK scorer\n#     cv=3,                # 3-fold cross-validation\n#     verbose=1            # Set to 0 for no output, or 1 for progress\n# )\n\n# grid_search.fit(X, y)\n# print(\"Best Parameters:\", grid_search.best_params_)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:51:36.19404Z","iopub.execute_input":"2024-12-19T09:51:36.194524Z","iopub.status.idle":"2024-12-19T09:51:36.202987Z","shell.execute_reply.started":"2024-12-19T09:51:36.194472Z","shell.execute_reply":"2024-12-19T09:51:36.20183Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"################\n# df = df.drop(columns=['id'])\n# # df = df.drop(columns=['sii'])\n# imputer = KNNImputer(n_neighbors=4)  # k=4\n# imputed_data = imputer.fit_transform(df)\n\n# df = pd.DataFrame(imputed_data, columns=df.columns)\n# df['id'] = train_ids\n# train = df\n####\nLight = LGBMRegressor(**LGBM_params, random_state=42, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n##########\nbest_weights = None\nbest_score = -float('inf')  # Assuming higher QWK score is better\ntrain = train.drop(columns=['id'])\n\n# Iterate over weights with step of 0.1\n# for i in np.arange(0.1, 1.1, 0.2):  # Loop for i\n#     for j in np.arange(0, 1.1, 0.2):  # Loop for j\n#         for k in np.arange(0, 1.1, 0.2):  # Loop for j\n\n#             print(f'lgbm weight: {i}, xgb wieight {j}, cat {k}')\n#             # Define the VotingRegressor with current weights\n#             voting_model = VotingRegressor(estimators=[\n#                 ('lightgbm', Light),\n#                 ('xgboost', XGB_Model),\n#                 ('catboost', CatBoost_Model)],\n#                 weights=[i, j, k]\n#             )\n    \n#             # Drop the 'id' column from training data\n    \n#             # Train the model and get predictions and QWK score\n#             vote_preds, score = TrainML(model_class=voting_model, test_data=test, acc=False)\n    \n#             # Track the best weights and score\n#             if score > best_score:\n#                 best_score = score\n#                 best_weights = [i, j, k]\n#                 print(f\"New Best Weights: {best_weights}, Score: {best_score}\")\n\n# # Print the final best weights and score\n# print(f\"Optimal Weights: {best_weights}, Best Score: {best_score}\")\n\nvoting_model = VotingRegressor(estimators=[\n                ('lightgbm', Light),\n                ('xgboost', XGB_Model),\n                ('catboost', CatBoost_Model)],\n                weights=[0.7, 0, 0.3]\n            )\n    \n            # Drop the 'id' column from training data\n    \n            # Train the model and get predictions and QWK score\nvote_preds, score = TrainML(model_class=voting_model, test_data=test, acc=False)\nvote_sub = pd.DataFrame({\n\n    'id': test['id'],\n    'sii': vote_preds\n})\nvote_sub.to_csv('submission.csv', index=False)\nvote_sub","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:51:36.206992Z","iopub.execute_input":"2024-12-19T09:51:36.20781Z","iopub.status.idle":"2024-12-19T09:52:44.722046Z","shell.execute_reply.started":"2024-12-19T09:51:36.207768Z","shell.execute_reply":"2024-12-19T09:52:44.720654Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"eq_model = VotingRegressor(estimators=[\n                ('lightgbm', Light),\n                ('xgboost', XGB_Model),\n                ('catboost', CatBoost_Model)],\n                weights=[1, 1, 1]\n            )\n    \n            # Drop the 'id' column from training data\n    \n            # Train the model and get predictions and QWK score\nvote_preds, score = TrainML(model_class=eq_model, test_data=test, acc=False)\neq = pd.DataFrame({\n\n    'id': test['id'],\n    'sii': vote_preds\n})\neq.to_csv('submission.csv', index=False)\neq\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:52:44.723474Z","iopub.execute_input":"2024-12-19T09:52:44.723842Z","iopub.status.idle":"2024-12-19T09:53:53.03904Z","shell.execute_reply.started":"2024-12-19T09:52:44.723806Z","shell.execute_reply":"2024-12-19T09:53:53.037824Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load and preprocess data\ndf = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\ndf = df.dropna(subset=['sii'])  # Keep labeled values only\ntest = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")\nseason_mapping = {\n    'Winter': -1,\n    'Spring': -0.5,\n    'Summer': 0.5,\n    'Fall': 1\n}\ndf = df.replace(season_mapping)\ntest = test.replace(season_mapping)\n\ntest_missing_columns = set(df.columns) - set(test.columns)\nfor col in test_missing_columns:\n    if col != 'sii':  # Retain the target column for training\n        df.drop(columns=col, inplace=True)\n\ntrain_ids = df['id']\ntest_ids = test['id']\ntrain_labels = df['sii']\ndf = df.drop(columns=['id'])\ndf = df.drop(columns=['sii'])\n\nimputer = KNNImputer(n_neighbors=4)\ndf_imputed = imputer.fit_transform(df)\ntest_imputed = imputer.transform(test.drop(columns=['id']))\n\nscaler = StandardScaler()\ndf_scaled = scaler.fit_transform(df_imputed)\ntest_scaled = scaler.transform(test_imputed)\n\nX = pd.DataFrame(df_scaled, columns=df.columns)\ny = train_labels\n\n# CNN model for numeric features\ndef build_cnn(input_shape):\n    model = Sequential([\n        Conv1D(filters=64, kernel_size=3, activation='relu', input_shape=input_shape),\n        MaxPooling1D(pool_size=2),\n        BatchNormalization(),\n        Dropout(0.3),\n        Conv1D(filters=128, kernel_size=3, activation='relu'),\n        MaxPooling1D(pool_size=2),\n        BatchNormalization(),\n        Dropout(0.3),\n        Flatten(),\n        Dense(128, activation='relu'),\n        Dropout(0.5),\n        Dense(1, activation='linear')  # Regression output\n    ])\n    model.compile(optimizer='adam', loss='mse', metrics=['mae'])\n    return model\n\n# Custom threshold rounding\ndef threshold_rounder(preds, thresholds):\n    return np.where(preds < thresholds[0], 0,\n                    np.where(preds < thresholds[1], 1,\n                             np.where(preds < thresholds[2], 2, 3)))\n\n# QWK evaluation\ndef evaluate_predictions(thresholds, y_true, preds):\n    rounded_preds = threshold_rounder(preds, thresholds)\n    return -cohen_kappa_score(y_true, rounded_preds, weights='quadratic')\n\n# Training\nn_splits = 5\nrandom_state = 42\nskf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=random_state)\ntest_preds = np.zeros((len(test_scaled), n_splits))\noof_preds = np.zeros(len(y))\n\nfor fold, (train_idx, val_idx) in enumerate(tqdm(skf.split(X, y), total=n_splits, desc=\"Training Folds\")):\n    X_train, X_val = X.iloc[train_idx].values, X.iloc[val_idx].values\n    y_train, y_val = y.iloc[train_idx].values, y.iloc[val_idx].values\n\n    # Reshape for Conv1D\n    X_train = X_train.reshape(-1, X_train.shape[1], 1)\n    X_val = X_val.reshape(-1, X_val.shape[1], 1)\n    test_reshaped = test_scaled.reshape(-1, test_scaled.shape[1], 1)\n\n    # Train CNN\n    cnn = build_cnn(input_shape=(X_train.shape[1], 1))\n    cnn.fit(X_train, y_train, validation_data=(X_val, y_val), epochs=10, batch_size=32, verbose=1)\n\n    # Train Random Forest\n    rf = RandomForestRegressor(n_estimators=100, random_state=random_state)\n    rf.fit(X_train.reshape(X_train.shape[0], -1), y_train)\n\n    # Combine predictions\n    cnn_val_preds = cnn.predict(X_val).flatten()\n    rf_val_preds = rf.predict(X_val.reshape(X_val.shape[0], -1))\n\n    val_preds = (cnn_val_preds + rf_val_preds) / 2\n\n    # Test predictions\n    cnn_test_preds = cnn.predict(test_reshaped).flatten()\n    rf_test_preds = rf.predict(test_scaled)\n\n    test_fold_preds = (cnn_test_preds + rf_test_preds) / 2\n\n    oof_preds[val_idx] = val_preds\n    test_preds[:, fold] = test_fold_preds\n\n    val_qwk = cohen_kappa_score(y_val, threshold_rounder(val_preds, [0.5, 1.5, 2.5]), weights='quadratic')\n    print(f\"Fold {fold + 1} - Validation QWK: {val_qwk:.4f}\")\n\n# Optimize thresholds\nopt_result = minimize(evaluate_predictions, [0.5, 1.5, 2.5], args=(y, oof_preds), method='Nelder-Mead')\noptimal_thresholds = opt_result.x\nfinal_qwk = cohen_kappa_score(y, threshold_rounder(oof_preds, optimal_thresholds), weights='quadratic')\nprint(f\"Optimized QWK: {final_qwk:.4f}\")\n\n# Test predictions\nfinal_test_preds = test_preds.mean(axis=1)\nfinal_test_preds = threshold_rounder(final_test_preds, optimal_thresholds)\n\nsubmission = pd.DataFrame({'id': test_ids, 'sii': final_test_preds})\nsubmission.to_csv('/kaggle/working/submission.csv', index=False)\nsubmission","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:53:53.041165Z","iopub.execute_input":"2024-12-19T09:53:53.041545Z","iopub.status.idle":"2024-12-19T09:55:48.781235Z","shell.execute_reply.started":"2024-12-19T09:53:53.041511Z","shell.execute_reply":"2024-12-19T09:55:48.780017Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n# Example dataframes\ndf_A = vote_sub # 0.7 0 0.3\ndf_B = eq # 1 1 1\ndf_C = submission # cnn\n\n# Merge dataframes on 'id'\nmerged_df = pd.merge(df_A, df_B, on='id', suffixes=('_A', '_B'))\nmerged_df = pd.merge(merged_df, df_C, on='id')\nmerged_df.rename(columns={'sii': 'sii_C'}, inplace=True)\n\n# Define weights\nweight_A = 0.49\nweight_B = 0.11\nweight_C = 0.40\n\n# Normalize weights\ntotal_weight = weight_A + weight_B + weight_C\nw_A = weight_A / total_weight\nw_B = weight_B / total_weight\nw_C = weight_C / total_weight\n\n# Compute weighted average\nmerged_df['sii'] = (\n    merged_df['sii_A'] * w_A + \n    merged_df['sii_B'] * w_B + \n    merged_df['sii_C'] * w_C\n).round()\n\ndf_sii = merged_df[['id', 'sii']].copy()\ndf_sii['sii'] = df_sii['sii'].astype(int)\ndf_sii.to_csv('submission.csv', index=False)\ndf_sii\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T10:09:15.702777Z","iopub.execute_input":"2024-12-19T10:09:15.70373Z","iopub.status.idle":"2024-12-19T10:09:15.729886Z","shell.execute_reply.started":"2024-12-19T10:09:15.703684Z","shell.execute_reply":"2024-12-19T10:09:15.728708Z"}},"outputs":[],"execution_count":null}]}