{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30805,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Imports and Set Up","metadata":{}},{"cell_type":"code","source":"!pip install torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cu118","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:04:16.295539Z","iopub.execute_input":"2024-12-11T00:04:16.296198Z","iopub.status.idle":"2024-12-11T00:04:25.602545Z","shell.execute_reply.started":"2024-12-11T00:04:16.296165Z","shell.execute_reply":"2024-12-11T00:04:25.601647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install pytorch-tabnet","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:04:25.604510Z","iopub.execute_input":"2024-12-11T00:04:25.604841Z","iopub.status.idle":"2024-12-11T00:04:34.208566Z","shell.execute_reply.started":"2024-12-11T00:04:25.604811Z","shell.execute_reply":"2024-12-11T00:04:34.207664Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport random\nimport warnings\nfrom concurrent.futures import ThreadPoolExecutor\nfrom sklearn.model_selection import train_test_split\nfrom scipy.optimize import minimize\nimport seaborn as sns\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nfrom lightgbm import LGBMClassifier, LGBMRegressor\nfrom matplotlib import pyplot as plt\nfrom sklearn.base import clone\nfrom sklearn.ensemble import VotingClassifier, VotingRegressor, StackingClassifier, StackingRegressor\nfrom sklearn.impute import KNNImputer\nfrom sklearn.metrics import (accuracy_score, cohen_kappa_score,\n                             confusion_matrix, f1_score, mean_absolute_error,\n                             mean_squared_error, precision_score, recall_score,\n                             classification_report, make_scorer)\nfrom sklearn.model_selection import (GridSearchCV, KFold, RandomizedSearchCV,\n                                     cross_val_score, ParameterGrid)\nfrom sklearn.svm import SVC\nfrom sklearn.linear_model import LogisticRegression, Ridge, Lasso\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.preprocessing import StandardScaler\nfrom tqdm import tqdm\nfrom xgboost import XGBClassifier, XGBRegressor\nfrom catboost import CatBoostClassifier, CatBoostRegressor\nfrom sklearn.ensemble import RandomForestClassifier, RandomForestRegressor\nfrom imblearn.under_sampling import NearMiss\nfrom pytorch_tabnet.tab_model import TabNetClassifier, TabNetRegressor","metadata":{"execution":{"iopub.status.busy":"2024-12-11T00:04:34.209905Z","iopub.execute_input":"2024-12-11T00:04:34.210200Z","iopub.status.idle":"2024-12-11T00:04:42.445930Z","shell.execute_reply.started":"2024-12-11T00:04:34.210171Z","shell.execute_reply":"2024-12-11T00:04:42.445143Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"warnings.filterwarnings('ignore')\npd.options.display.max_columns = None","metadata":{"execution":{"iopub.status.busy":"2024-12-11T00:04:42.448460Z","iopub.execute_input":"2024-12-11T00:04:42.449265Z","iopub.status.idle":"2024-12-11T00:04:42.454400Z","shell.execute_reply.started":"2024-12-11T00:04:42.449217Z","shell.execute_reply":"2024-12-11T00:04:42.453472Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def set_seed(seed_value=2024):\n    random.seed(seed_value)\n    np.random.seed(seed_value)\n    torch.manual_seed(seed_value)\n    torch.cuda.manual_seed(seed_value)\n    torch.backends.cudnn.deterministic = True\n\nset_seed(2024)","metadata":{"execution":{"iopub.status.busy":"2024-12-11T00:04:42.455788Z","iopub.execute_input":"2024-12-11T00:04:42.456148Z","iopub.status.idle":"2024-12-11T00:04:42.472143Z","shell.execute_reply.started":"2024-12-11T00:04:42.456105Z","shell.execute_reply":"2024-12-11T00:04:42.471129Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Processing","metadata":{}},{"cell_type":"markdown","source":"### Load in Files","metadata":{}},{"cell_type":"code","source":"file_path = '/kaggle/input/child-mind-institute-problematic-internet-use'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:04:42.473145Z","iopub.execute_input":"2024-12-11T00:04:42.473391Z","iopub.status.idle":"2024-12-11T00:04:42.477426Z","shell.execute_reply.started":"2024-12-11T00:04:42.473367Z","shell.execute_reply":"2024-12-11T00:04:42.476785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAIN_CSV = f'{file_path}/train.csv'\nTEST_CSV = f'{file_path}/test.csv'\nSAMPLE_SUBMISSION_CSV = f'{file_path}/sample_submission.csv'\nSERIES_TRAIN_DIR = f'{file_path}/series_train.parquet'\nSERIES_TEST_DIR = f'{file_path}/series_test.parquet'\n\n\ntrain_df = pd.read_csv(TRAIN_CSV)\ntest_df = pd.read_csv(TEST_CSV) # \nsample_submission_df = pd.read_csv(SAMPLE_SUBMISSION_CSV)\n\n# Drop all the PCIAT variables as they are not present in the test data\nfor col in train_df.columns:\n    if 'PCIAT' in col:\n        train_df.drop(col, axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-12-11T00:04:42.478452Z","iopub.execute_input":"2024-12-11T00:04:42.479087Z","iopub.status.idle":"2024-12-11T00:04:42.582052Z","shell.execute_reply.started":"2024-12-11T00:04:42.479048Z","shell.execute_reply":"2024-12-11T00:04:42.581324Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to process individual time series files\ndef process_time_series(file_name, directory):\n    df = pd.read_parquet(os.path.join(directory, file_name, 'part-0.parquet'))\n    df = df.drop('step', axis=1)\n    stats = df.describe().values.flatten()\n    record_id = file_name.split('=')[1]\n    return stats, record_id","metadata":{"execution":{"iopub.status.busy":"2024-12-11T00:04:42.583123Z","iopub.execute_input":"2024-12-11T00:04:42.583377Z","iopub.status.idle":"2024-12-11T00:04:42.588799Z","shell.execute_reply.started":"2024-12-11T00:04:42.583352Z","shell.execute_reply":"2024-12-11T00:04:42.587917Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to load and aggregate time series data\ndef load_time_series_data(directory):\n    file_names = os.listdir(directory)\n    stats_list = []\n    ids_list = []\n\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_time_series(fname, directory), file_names),\n                            total=len(file_names)))\n\n    for stats, record_id in results:\n        stats_list.append(stats)\n        ids_list.append(record_id)\n\n    stats_df = pd.DataFrame(stats_list, columns=[f'stat_{i}' for i in range(len(stats_list[0]))])\n    stats_df['id'] = ids_list\n    return stats_df","metadata":{"execution":{"iopub.status.busy":"2024-12-11T00:04:42.590047Z","iopub.execute_input":"2024-12-11T00:04:42.590422Z","iopub.status.idle":"2024-12-11T00:04:42.598871Z","shell.execute_reply.started":"2024-12-11T00:04:42.590381Z","shell.execute_reply":"2024-12-11T00:04:42.597960Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_series_df = load_time_series_data(SERIES_TRAIN_DIR)\n# test_series_df = load_time_series_data(SERIES_TEST_DIR)","metadata":{"execution":{"iopub.status.busy":"2024-12-11T00:04:42.601939Z","iopub.execute_input":"2024-12-11T00:04:42.602181Z","iopub.status.idle":"2024-12-11T00:05:54.931286Z","shell.execute_reply.started":"2024-12-11T00:04:42.602159Z","shell.execute_reply":"2024-12-11T00:05:54.930394Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Replace with missing for the split and then reassign the values at the end\ntrain_df['sii'] = train_df['sii'].replace({np.nan: -1})\ntrain_df, test_df = train_test_split(train_df, test_size=0.2, random_state=2024, stratify=train_df['sii'])\ntest_series_df = train_series_df[train_series_df.id.isin(test_df.id)]\ntrain_series_df = train_series_df[train_series_df.id.isin(train_df.id)]\ntrain_df['sii'] = train_df['sii'].replace({-1: np.nan})\ntest_df['sii'] = test_df['sii'].replace({-1: np.nan})\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:05:54.932615Z","iopub.execute_input":"2024-12-11T00:05:54.933371Z","iopub.status.idle":"2024-12-11T00:05:54.952083Z","shell.execute_reply.started":"2024-12-11T00:05:54.933330Z","shell.execute_reply":"2024-12-11T00:05:54.951336Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_series_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:05:54.953271Z","iopub.execute_input":"2024-12-11T00:05:54.953854Z","iopub.status.idle":"2024-12-11T00:05:55.036826Z","shell.execute_reply.started":"2024-12-11T00:05:54.953815Z","shell.execute_reply":"2024-12-11T00:05:55.035971Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_series_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:05:55.038106Z","iopub.execute_input":"2024-12-11T00:05:55.038482Z","iopub.status.idle":"2024-12-11T00:05:55.122684Z","shell.execute_reply.started":"2024-12-11T00:05:55.038442Z","shell.execute_reply":"2024-12-11T00:05:55.121793Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Encoding of Time Series Data","metadata":{}},{"cell_type":"code","source":"class TimeSeriesAutoencoder(nn.Module):\n    def __init__(self, input_size, encoding_size):\n        super(TimeSeriesAutoencoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_size, encoding_size * 3),\n            nn.ReLU(),\n            nn.Linear(encoding_size * 3, encoding_size * 2),\n            nn.ReLU(),\n            nn.Linear(encoding_size * 2, encoding_size),\n            nn.ReLU()\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_size, input_size * 2),\n            nn.ReLU(),\n            nn.Linear(input_size * 2, input_size * 3),\n            nn.ReLU(),\n            nn.Linear(input_size * 3, input_size),\n            nn.Sigmoid()\n        )\n\n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded","metadata":{"execution":{"iopub.status.busy":"2024-12-11T00:05:55.123840Z","iopub.execute_input":"2024-12-11T00:05:55.124107Z","iopub.status.idle":"2024-12-11T00:05:55.130080Z","shell.execute_reply.started":"2024-12-11T00:05:55.124083Z","shell.execute_reply":"2024-12-11T00:05:55.129227Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def encodedDataPrep(train, test):\n    scaler = StandardScaler()\n    scaled_data_train = scaler.fit_transform(train.drop('id', axis=1))\n    scaled_data_test = scaler.transform(test.drop('id', axis=1))\n    tensor_data_train = torch.FloatTensor(scaled_data_train)\n    tensor_data_test = torch.FloatTensor(scaled_data_train)\n    return tensor_data_train, tensor_data_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:05:55.131321Z","iopub.execute_input":"2024-12-11T00:05:55.132218Z","iopub.status.idle":"2024-12-11T00:05:55.142960Z","shell.execute_reply.started":"2024-12-11T00:05:55.132177Z","shell.execute_reply":"2024-12-11T00:05:55.142077Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_encoded_features(tensor_data, train=True, autoencoder=None, encoding_dim=60, epochs=100, batch_size=32):\n    input_dim = tensor_data.shape[1]\n    autoencoder = TimeSeriesAutoencoder(input_dim, encoding_dim)\n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(autoencoder.parameters())\n    for epoch in range(epochs):\n        for i in range(0, len(tensor_data), batch_size):\n            batch = tensor_data[i:i + batch_size]\n            optimizer.zero_grad()\n            outputs = autoencoder(batch)\n            loss = criterion(outputs, batch)\n            loss.backward()\n            optimizer.step()\n\n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}')\n\n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(tensor_data).numpy()\n\n    encoded_df = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n    return encoded_df, autoencoder","metadata":{"execution":{"iopub.status.busy":"2024-12-11T00:05:55.144114Z","iopub.execute_input":"2024-12-11T00:05:55.144531Z","iopub.status.idle":"2024-12-11T00:05:55.157067Z","shell.execute_reply.started":"2024-12-11T00:05:55.144492Z","shell.execute_reply":"2024-12-11T00:05:55.156238Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def encode_test_data(autoencoder, test_tensor):\n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(test_tensor).numpy()\n    encoded_df = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n    return encoded_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:05:55.158222Z","iopub.execute_input":"2024-12-11T00:05:55.158834Z","iopub.status.idle":"2024-12-11T00:05:55.170316Z","shell.execute_reply.started":"2024-12-11T00:05:55.158795Z","shell.execute_reply":"2024-12-11T00:05:55.169505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tensor_data_train, tensor_data_test = encodedDataPrep(train_series_df, test_series_df)\ntrain_encoded, autoencoder = get_encoded_features(tensor_data_train)\ntest_encoded = encode_test_data(autoencoder, tensor_data_test)\n\ntrain_encoded['id'] = train_series_df['id']\ntest_encoded['id'] = test_series_df['id']\n\ntrain_df = train_df.merge(train_encoded, on='id', how='left')\ntest_df = test_df.merge(test_encoded, on='id', how='left')","metadata":{"execution":{"iopub.status.busy":"2024-12-11T00:05:55.171287Z","iopub.execute_input":"2024-12-11T00:05:55.171537Z","iopub.status.idle":"2024-12-11T00:06:04.096504Z","shell.execute_reply.started":"2024-12-11T00:05:55.171512Z","shell.execute_reply":"2024-12-11T00:06:04.095511Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Imputation of Missing Numerical Values","metadata":{}},{"cell_type":"code","source":"# Imputing missing values using KNN imputer\n\ndef impute_missing_values(train_df, test_df, n_neighbors=5):\n    # Select only numeric columns\n    numeric_columns = train_df.select_dtypes(include=['float64', 'float32', 'int64']).columns\n\n    # Initialize the KNNImputer\n    imputer = KNNImputer(n_neighbors=n_neighbors)\n\n    # Fit on the training dataset and transform both datasets\n    train_imputed_array = imputer.fit_transform(train_df[numeric_columns])\n    test_imputed_array = imputer.transform(test_df[numeric_columns])\n\n    # Create DataFrames from the imputed arrays\n    train_imputed = pd.DataFrame(train_imputed_array, columns=numeric_columns, index=train_df.index)\n    test_imputed = pd.DataFrame(test_imputed_array, columns=numeric_columns, index=test_df.index)\n\n    # Replace numeric columns with imputed data\n    train_df[numeric_columns] = train_imputed\n    test_df[numeric_columns] = test_imputed\n\n    return train_df, test_df","metadata":{"execution":{"iopub.status.busy":"2024-12-11T00:06:04.097802Z","iopub.execute_input":"2024-12-11T00:06:04.098847Z","iopub.status.idle":"2024-12-11T00:06:04.104703Z","shell.execute_reply.started":"2024-12-11T00:06:04.098806Z","shell.execute_reply":"2024-12-11T00:06:04.103764Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train_df.replace([np.inf, -np.inf], np.nan, inplace=True) # Debug why we have inf values\n\n# Apply imputation after splitting the dataset\ntrain_df, test_df = impute_missing_values(train_df, test_df)\n\n# Ensure target column 'sii' remains integer after imputation\ntrain_df['sii'] = train_df['sii'].round().astype(int)\ntest_df['sii'] = test_df['sii'].round().astype(int)\n\n# Imputation is needed in test set for some cases but not others to revisit\n\nprint(train_df.isna().sum())\nprint(test_df.isna().sum())","metadata":{"execution":{"iopub.status.busy":"2024-12-11T00:06:04.105814Z","iopub.execute_input":"2024-12-11T00:06:04.106098Z","iopub.status.idle":"2024-12-11T00:06:12.885999Z","shell.execute_reply.started":"2024-12-11T00:06:04.106068Z","shell.execute_reply":"2024-12-11T00:06:12.884867Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Categorical Processing","metadata":{}},{"cell_type":"code","source":"categorical_cols = list(train_df.select_dtypes(include=['object']).columns) #sii (outcome var) is categorical but we are encoding that differently\ncategorical_cols.remove('id')\nprint(categorical_cols)\n\ndef preprocess_categorical(df):\n    for col in categorical_cols:\n        df[col] = df[col].fillna('Missing').astype('category')\n    return df\n\ntrain_df = preprocess_categorical(train_df)\ntest_df = preprocess_categorical(test_df)\n\ntrain_df = pd.get_dummies(train_df, columns = categorical_cols, drop_first=True, dtype='int')\ntest_df = pd.get_dummies(test_df, columns = categorical_cols, drop_first=True, dtype='int')","metadata":{"execution":{"iopub.status.busy":"2024-12-11T00:06:12.887238Z","iopub.execute_input":"2024-12-11T00:06:12.887557Z","iopub.status.idle":"2024-12-11T00:06:12.950566Z","shell.execute_reply.started":"2024-12-11T00:06:12.887526Z","shell.execute_reply":"2024-12-11T00:06:12.949660Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = train_df.drop('id', axis=1)\ntest_df = test_df.drop('id', axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-12-11T00:06:12.951490Z","iopub.execute_input":"2024-12-11T00:06:12.951733Z","iopub.status.idle":"2024-12-11T00:06:12.958860Z","shell.execute_reply.started":"2024-12-11T00:06:12.951709Z","shell.execute_reply":"2024-12-11T00:06:12.957889Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the Quadratic Weighted Kappa metric\ndef quadratic_weighted_kappa(y_actual, y_predicted):\n    return cohen_kappa_score(y_actual, y_predicted, weights='quadratic')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:06:12.960062Z","iopub.execute_input":"2024-12-11T00:06:12.960359Z","iopub.status.idle":"2024-12-11T00:06:12.967893Z","shell.execute_reply.started":"2024-12-11T00:06:12.960332Z","shell.execute_reply":"2024-12-11T00:06:12.967011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def make_cm(y_true, y_pred, title):\n    cm = confusion_matrix(y_true, y_pred)\n    plt.figure(figsize=(6, 5))\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=np.unique(y_test), yticklabels=np.unique(y_test))\n    plt.xlabel('Predicted')\n    plt.ylabel('True')\n    plt.title(f'{title} Confusion Matrix')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:06:12.969143Z","iopub.execute_input":"2024-12-11T00:06:12.969488Z","iopub.status.idle":"2024-12-11T00:06:12.981951Z","shell.execute_reply.started":"2024-12-11T00:06:12.969459Z","shell.execute_reply":"2024-12-11T00:06:12.981045Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Resampling of Training Set","metadata":{}},{"cell_type":"code","source":"X = train_df.drop(columns='sii', axis=1)\ny = train_df['sii'].astype(int)\n\n# # Apply NearMiss for downsampling\n# nm = NearMiss(sampling_strategy='majority')  # You can change version to 2 or 3\n# # These are global variables for now\n# X_res, y_res = nm.fit_resample(X, y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:06:12.983016Z","iopub.execute_input":"2024-12-11T00:06:12.983312Z","iopub.status.idle":"2024-12-11T00:06:12.995672Z","shell.execute_reply.started":"2024-12-11T00:06:12.983285Z","shell.execute_reply":"2024-12-11T00:06:12.994885Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Baseline - Random Forest Model","metadata":{}},{"cell_type":"code","source":"def train_and_evaluate(model, param_dist, n_iter=10, eval_metric=None):\n    kf = StratifiedKFold(n_splits=3, shuffle=True, random_state=42)\n    search = RandomizedSearchCV(model, param_dist, n_iter=n_iter, cv=kf, n_jobs=-1, random_state=42, verbose=2, scoring=eval_metric)\n    search.fit(X, y) # The resampling did not help\n    best_model = search.best_estimator_\n    y_pred = best_model.predict(X)\n    print(f\"Best Parameters: {search.best_params_}\")\n    return best_model, y_pred, y","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:06:12.996988Z","iopub.execute_input":"2024-12-11T00:06:12.997288Z","iopub.status.idle":"2024-12-11T00:06:13.005224Z","shell.execute_reply.started":"2024-12-11T00:06:12.997258Z","shell.execute_reply":"2024-12-11T00:06:13.004242Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rf_param_dist = {\n    'n_estimators': [100, 200, 300],\n    'max_depth': [3, 5],\n    'min_samples_split': [2, 5],\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:06:13.006885Z","iopub.execute_input":"2024-12-11T00:06:13.007352Z","iopub.status.idle":"2024-12-11T00:06:13.017763Z","shell.execute_reply.started":"2024-12-11T00:06:13.007305Z","shell.execute_reply":"2024-12-11T00:06:13.016756Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Classifier","metadata":{}},{"cell_type":"code","source":"# Training and evaluation function\ndef train_and_evaluate_classifier(model, param_dist, eval_metric=None, n_iter=10):\n    best_model, y_pred, y = train_and_evaluate(model, param_dist, n_iter, eval_metric)\n    train_kappa = quadratic_weighted_kappa(y, y_pred)\n    train_accuracy = accuracy_score(y, y_pred)\n    train_f1 = f1_score(y, y_pred, average='weighted')\n    print(f\"Best Train QWK: {train_kappa:.4f}\")\n    print(f\"Best Train Accuracy: {train_accuracy:.4f}\")\n    print(f\"Best Train F1: {train_f1:.4f}\")\n    return best_model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:06:13.023659Z","iopub.execute_input":"2024-12-11T00:06:13.024035Z","iopub.status.idle":"2024-12-11T00:06:13.032843Z","shell.execute_reply.started":"2024-12-11T00:06:13.024006Z","shell.execute_reply":"2024-12-11T00:06:13.031888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"RF = RandomForestClassifier(random_state=42)\nbest_model = train_and_evaluate_classifier(RF, rf_param_dist, n_iter=3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:06:13.034001Z","iopub.execute_input":"2024-12-11T00:06:13.034354Z","iopub.status.idle":"2024-12-11T00:06:19.340504Z","shell.execute_reply.started":"2024-12-11T00:06:13.034323Z","shell.execute_reply":"2024-12-11T00:06:19.339428Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = test_df.drop('sii', axis=1)\ny_test = test_df['sii'].astype(int)\ny_test_pred = best_model.predict(X_test)\ntest_kappa = quadratic_weighted_kappa(y_test, y_test_pred)\ntest_accuracy = accuracy_score(y_test, y_test_pred)\ntest_f1 = f1_score(y_test, y_test_pred, average='weighted')\nprint(f\"Test QWK: {Fore.GREEN}{Style.BRIGHT}{test_kappa:.4f}{Style.RESET_ALL}\")\nprint(f\"Test Accuracy: {Fore.GREEN}{Style.BRIGHT}{test_accuracy:.4f}{Style.RESET_ALL}\")\nprint(f\"Test F1: {Fore.GREEN}{Style.BRIGHT}{test_f1:.4f}{Style.RESET_ALL}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:06:19.341825Z","iopub.execute_input":"2024-12-11T00:06:19.342115Z","iopub.status.idle":"2024-12-11T00:06:19.378925Z","shell.execute_reply.started":"2024-12-11T00:06:19.342085Z","shell.execute_reply":"2024-12-11T00:06:19.378046Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Regression","metadata":{}},{"cell_type":"code","source":"# Function to apply thresholds to continuous predictions\ndef apply_thresholds(predictions, thresholds):\n    return np.digitize(predictions, bins=thresholds)\n\n# Function to optimize thresholds to maximize QWK\ndef optimize_thresholds(y_true, predictions):\n    def loss_func(thresh):\n        # Ensure thresholds are sorted\n        thresh_sorted = np.sort(thresh)\n        preds = apply_thresholds(predictions, thresh_sorted)\n        return -quadratic_weighted_kappa(y_true, preds)\n    \n    initial_thresholds = [0.5, 1.5, 2.5]  # Initial guesses for thresholds\n    bounds = [(0, 3)] * 3  # Assuming classes are 0,1,2,3\n    result = minimize(loss_func, initial_thresholds, method='Nelder-Mead')\n    return result.x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:06:19.380264Z","iopub.execute_input":"2024-12-11T00:06:19.381013Z","iopub.status.idle":"2024-12-11T00:06:19.386943Z","shell.execute_reply.started":"2024-12-11T00:06:19.380964Z","shell.execute_reply":"2024-12-11T00:06:19.385913Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_and_evaluate_regressor(model, param_dist, n_iter=10, eval_metric=None):\n    best_model, y_pred, y = train_and_evaluate(model, param_dist, n_iter, eval_metric)\n    thresholds = optimize_thresholds(y, y_pred)\n    y_pred = apply_thresholds(y_pred, thresholds)\n    train_kappa = quadratic_weighted_kappa(y, y_pred)\n    train_accuracy = accuracy_score(y, y_pred)\n    train_f1 = f1_score(y, y_pred, average='weighted')\n    print(f\"Best Train QWK: {train_kappa:.4f}\")\n    print(f\"Best Train Accuracy: {train_accuracy:.4f}\")\n    print(f\"Best Train F1: {train_f1:.4f}\")\n    return best_model, thresholds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:06:19.388181Z","iopub.execute_input":"2024-12-11T00:06:19.388545Z","iopub.status.idle":"2024-12-11T00:06:19.397678Z","shell.execute_reply.started":"2024-12-11T00:06:19.388496Z","shell.execute_reply":"2024-12-11T00:06:19.396878Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"RF = RandomForestRegressor(random_state=42)\nbest_model, thresholds = train_and_evaluate_regressor(RF, rf_param_dist, n_iter=3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:06:19.398835Z","iopub.execute_input":"2024-12-11T00:06:19.399648Z","iopub.status.idle":"2024-12-11T00:06:52.061291Z","shell.execute_reply.started":"2024-12-11T00:06:19.399606Z","shell.execute_reply":"2024-12-11T00:06:52.060336Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = test_df.drop('sii', axis=1)\ny_test = test_df['sii'].astype(int)\ntest_X = X_test.values.astype(np.float32)\nraw_preds = best_model.predict(test_X)\ny_test_pred = apply_thresholds(raw_preds, thresholds)\ntest_kappa = quadratic_weighted_kappa(y_test, y_test_pred)\ntest_accuracy = accuracy_score(y_test, y_test_pred)\ntest_f1 = f1_score(y_test, y_test_pred, average='weighted')\n\nprint(f\"Best Test QWK: {test_kappa:.4f}\")\nprint(f\"Best Train Accuracy: {test_accuracy:.4f}\")\nprint(f\"Best Train F1: {test_f1:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:06:52.062362Z","iopub.execute_input":"2024-12-11T00:06:52.062649Z","iopub.status.idle":"2024-12-11T00:06:52.087278Z","shell.execute_reply.started":"2024-12-11T00:06:52.062623Z","shell.execute_reply":"2024-12-11T00:06:52.086434Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model 1 - Voting Classifier","metadata":{}},{"cell_type":"code","source":"# Training and evaluation function\ndef train_and_evaluate_classifier(model, param_dist, eval_metric=None, n_iter=10):\n    best_model, y_pred, y = train_and_evaluate(model, param_dist, n_iter, eval_metric)\n    train_kappa = quadratic_weighted_kappa(y, y_pred)\n    train_accuracy = accuracy_score(y, y_pred)\n    train_f1 = f1_score(y, y_pred, average='weighted')\n    print(f\"Best Train QWK: {train_kappa:.4f}\")\n    print(f\"Best Train Accuracy: {train_accuracy:.4f}\")\n    print(f\"Best Train F1: {train_f1:.4f}\")\n    return best_model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:06:52.088320Z","iopub.execute_input":"2024-12-11T00:06:52.088618Z","iopub.status.idle":"2024-12-11T00:06:52.093489Z","shell.execute_reply.started":"2024-12-11T00:06:52.088567Z","shell.execute_reply":"2024-12-11T00:06:52.092632Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the classifiers\nrf = RandomForestClassifier(random_state=42)\nsvc = SVC(probability=True, random_state=42)\nxgb_model = XGBClassifier(random_state=42, objective='multi:softprob', num_class=4, verbosity=0, learning_rate=0.05)\nlgb_model = LGBMClassifier(random_state=42, objective='multiclass', num_class=4, verbose=-1, learning_rate=0.05)\ncatb_model = CatBoostClassifier(random_state=42, objective='MultiClass', verbose=False, learning_rate=0.05)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:06:52.094465Z","iopub.execute_input":"2024-12-11T00:06:52.094733Z","iopub.status.idle":"2024-12-11T00:06:52.107037Z","shell.execute_reply.started":"2024-12-11T00:06:52.094708Z","shell.execute_reply":"2024-12-11T00:06:52.106264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"param_dist = {\n            # 'rf__n_estimators': [100, 200, 300], 'rf__max_depth': [3], 'rf__min_samples_split': [2, 5, 10], 'rf__min_samples_leaf': [1], 'rf__max_features': ['auto', 'sqrt', 'log2'],\n            #   'svc__C': [0.1, 1, 10], 'svc__gamma': ['scale', 'auto'], 'svc__kernel': ['linear', 'rbf'],\n              'xgb__n_estimators': [200], 'xgb__max_depth': [2, 3],\n              'lgb__n_estimators': [100], 'lgb__num_leaves': [5, 10],\n              'cat__iterations': [100, 200], 'cat__depth': [2, 4]\n              }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:06:52.108062Z","iopub.execute_input":"2024-12-11T00:06:52.108309Z","iopub.status.idle":"2024-12-11T00:06:52.119348Z","shell.execute_reply.started":"2024-12-11T00:06:52.108285Z","shell.execute_reply":"2024-12-11T00:06:52.118534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Create an ensemble using VotingClassifier with soft voting\nensemble_model = VotingClassifier(\n    estimators=[('lgb', lgb_model), \n                ('xgb', xgb_model), \n                ('cat', catb_model)\n                # ('rf', rf) \n                # ,('svc', svc)\n                ],\n    voting='soft',\n    # weights=[4.0, 4.0, 5.0],\n    n_jobs=-1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:06:52.120382Z","iopub.execute_input":"2024-12-11T00:06:52.120724Z","iopub.status.idle":"2024-12-11T00:06:52.128684Z","shell.execute_reply.started":"2024-12-11T00:06:52.120688Z","shell.execute_reply":"2024-12-11T00:06:52.128009Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_model = train_and_evaluate_classifier(ensemble_model, param_dist, n_iter=2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:06:52.129653Z","iopub.execute_input":"2024-12-11T00:06:52.129910Z","iopub.status.idle":"2024-12-11T00:07:13.074571Z","shell.execute_reply.started":"2024-12-11T00:06:52.129885Z","shell.execute_reply":"2024-12-11T00:07:13.073667Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = test_df.drop('sii', axis=1)\ny_test = test_df['sii'].astype(int)\ny_test_pred = best_model.predict(X_test)\ntest_kappa = quadratic_weighted_kappa(y_test, y_test_pred)\ntest_accuracy = accuracy_score(y_test, y_test_pred)\ntest_f1 = f1_score(y_test, y_test_pred, average='weighted')\nprint(f\"Test QWK: {Fore.GREEN}{Style.BRIGHT}{test_kappa:.4f}{Style.RESET_ALL}\")\nprint(f\"Test Accuracy: {Fore.GREEN}{Style.BRIGHT}{test_accuracy:.4f}{Style.RESET_ALL}\")\nprint(f\"Test F1: {Fore.GREEN}{Style.BRIGHT}{test_f1:.4f}{Style.RESET_ALL}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:07:13.075762Z","iopub.execute_input":"2024-12-11T00:07:13.076135Z","iopub.status.idle":"2024-12-11T00:07:13.142749Z","shell.execute_reply.started":"2024-12-11T00:07:13.076092Z","shell.execute_reply":"2024-12-11T00:07:13.141616Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"make_cm(y_test, y_test_pred, \"Voting Classifier - Test\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:07:13.144361Z","iopub.execute_input":"2024-12-11T00:07:13.144834Z","iopub.status.idle":"2024-12-11T00:07:13.500514Z","shell.execute_reply.started":"2024-12-11T00:07:13.144783Z","shell.execute_reply":"2024-12-11T00:07:13.499619Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model 1.5 - Stacking Classifier","metadata":{}},{"cell_type":"code","source":"\n# Create an ensemble using VotingClassifier with soft voting\nensemble_model = StackingClassifier(\n    estimators=[('lgb', lgb_model), \n                ('xgb', xgb_model), \n                ('cat', catb_model)\n                # ('rf', rf) \n                # ,('svc', svc)\n                ],\n    # weights=[4.0, 4.0, 5.0],\n    n_jobs=-1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:07:13.501902Z","iopub.execute_input":"2024-12-11T00:07:13.502286Z","iopub.status.idle":"2024-12-11T00:07:13.507290Z","shell.execute_reply.started":"2024-12-11T00:07:13.502246Z","shell.execute_reply":"2024-12-11T00:07:13.506367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_model = train_and_evaluate_classifier(ensemble_model, param_dist, n_iter=2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:07:13.508633Z","iopub.execute_input":"2024-12-11T00:07:13.508980Z","iopub.status.idle":"2024-12-11T00:08:42.775396Z","shell.execute_reply.started":"2024-12-11T00:07:13.508943Z","shell.execute_reply":"2024-12-11T00:08:42.774034Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = test_df.drop('sii', axis=1)\ny_test = test_df['sii'].astype(int)\ny_test_pred = best_model.predict(X_test)\ntest_kappa = quadratic_weighted_kappa(y_test, y_test_pred)\ntest_accuracy = accuracy_score(y_test, y_test_pred)\ntest_f1 = f1_score(y_test, y_test_pred, average='weighted')\nprint(f\"Test QWK: {Fore.GREEN}{Style.BRIGHT}{test_kappa:.4f}{Style.RESET_ALL}\")\nprint(f\"Test Accuracy: {Fore.GREEN}{Style.BRIGHT}{test_accuracy:.4f}{Style.RESET_ALL}\")\nprint(f\"Test F1: {Fore.GREEN}{Style.BRIGHT}{test_f1:.4f}{Style.RESET_ALL}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:08:42.776920Z","iopub.execute_input":"2024-12-11T00:08:42.777428Z","iopub.status.idle":"2024-12-11T00:08:42.821695Z","shell.execute_reply.started":"2024-12-11T00:08:42.777384Z","shell.execute_reply":"2024-12-11T00:08:42.820571Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"make_cm(y_test, y_test_pred, \"Stacking Classifier - Test\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:08:42.822999Z","iopub.execute_input":"2024-12-11T00:08:42.823401Z","iopub.status.idle":"2024-12-11T00:08:43.106345Z","shell.execute_reply.started":"2024-12-11T00:08:42.823359Z","shell.execute_reply":"2024-12-11T00:08:43.105467Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model 2 - Voting Regressor","metadata":{}},{"cell_type":"code","source":"# Function to apply thresholds to continuous predictions\ndef apply_thresholds(predictions, thresholds):\n    return np.digitize(predictions, bins=thresholds)\n\n# Function to optimize thresholds to maximize QWK\ndef optimize_thresholds(y_true, predictions):\n    def loss_func(thresh):\n        # Ensure thresholds are sorted\n        thresh_sorted = np.sort(thresh)\n        preds = apply_thresholds(predictions, thresh_sorted)\n        return -quadratic_weighted_kappa(y_true, preds)\n    \n    initial_thresholds = [0.5, 1.5, 2.5]  # Initial guesses for thresholds\n    bounds = [(0, 3)] * 3  # Assuming classes are 0,1,2,3\n    result = minimize(loss_func, initial_thresholds, method='Nelder-Mead')\n    return result.x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:08:43.107408Z","iopub.execute_input":"2024-12-11T00:08:43.107696Z","iopub.status.idle":"2024-12-11T00:08:43.113252Z","shell.execute_reply.started":"2024-12-11T00:08:43.107670Z","shell.execute_reply":"2024-12-11T00:08:43.112394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_and_evaluate_regressor(model, param_dist, n_iter=10, eval_metric=None):\n    best_model, y_pred, y = train_and_evaluate(model, param_dist, n_iter, eval_metric)\n    thresholds = optimize_thresholds(y, y_pred)\n    y_pred = apply_thresholds(y_pred, thresholds)\n    train_kappa = quadratic_weighted_kappa(y, y_pred)\n    train_accuracy = accuracy_score(y, y_pred)\n    train_f1 = f1_score(y, y_pred, average='weighted')\n    print(f\"Best Train QWK: {train_kappa:.4f}\")\n    print(f\"Best Train Accuracy: {train_accuracy:.4f}\")\n    print(f\"Best Train F1: {train_f1:.4f}\")\n    return best_model, thresholds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:08:43.114242Z","iopub.execute_input":"2024-12-11T00:08:43.114508Z","iopub.status.idle":"2024-12-11T00:08:43.123870Z","shell.execute_reply.started":"2024-12-11T00:08:43.114468Z","shell.execute_reply":"2024-12-11T00:08:43.123074Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"svm = SVC(random_state=42)\nxgb_model = XGBRegressor(random_state=42, verbosity=0, learning_rate=0.05)\nlgb_model = LGBMRegressor(random_state=42, verbose=-1, learning_rate=0.05)\ncatb_model = CatBoostRegressor(random_state=42, verbose=False, learning_rate=0.05)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:08:43.124712Z","iopub.execute_input":"2024-12-11T00:08:43.124949Z","iopub.status.idle":"2024-12-11T00:08:43.138317Z","shell.execute_reply.started":"2024-12-11T00:08:43.124925Z","shell.execute_reply":"2024-12-11T00:08:43.137557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"param_dist = {\n            # 'rf__n_estimators': [100, 200, 300], 'rf__max_depth': [3], 'rf__min_samples_split': [2, 5, 10], 'rf__min_samples_leaf': [1], 'rf__max_features': ['auto', 'sqrt', 'log2'],\n            #   'svc__C': [0.1, 1, 10], 'svc__gamma': ['scale', 'auto'], 'svc__kernel': ['linear', 'rbf'],\n              'xgb__n_estimators': [200], 'xgb__max_depth': [2, 3],\n              'lgb__n_estimators': [100], 'lgb__num_leaves': [5, 10],\n              'cat__iterations': [100, 200], 'cat__depth': [2, 4]\n              }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:08:43.139159Z","iopub.execute_input":"2024-12-11T00:08:43.139402Z","iopub.status.idle":"2024-12-11T00:08:43.148092Z","shell.execute_reply.started":"2024-12-11T00:08:43.139377Z","shell.execute_reply":"2024-12-11T00:08:43.147434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create an ensemble using VotingClassifier with soft voting\nensemble_model = VotingRegressor(\n    estimators=[('lgb', lgb_model), \n                ('xgb', xgb_model), \n                ('cat', catb_model)\n                # ('rf', rf) \n                # ,('svc', svc)\n                ],\n    # weights=[4.0, 4.0, 5.0],\n    n_jobs=-1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:08:43.149078Z","iopub.execute_input":"2024-12-11T00:08:43.149390Z","iopub.status.idle":"2024-12-11T00:08:43.159282Z","shell.execute_reply.started":"2024-12-11T00:08:43.149352Z","shell.execute_reply":"2024-12-11T00:08:43.158599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_model, thresholds = train_and_evaluate_regressor(ensemble_model, param_dist, n_iter=2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:08:43.160249Z","iopub.execute_input":"2024-12-11T00:08:43.160563Z","iopub.status.idle":"2024-12-11T00:08:49.766079Z","shell.execute_reply.started":"2024-12-11T00:08:43.160515Z","shell.execute_reply":"2024-12-11T00:08:49.765163Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = test_df.drop('sii', axis=1)\ny_test = test_df['sii'].astype(int)\ny_test_pred = best_model.predict(X_test)\ny_test_pred = apply_thresholds(y_test_pred, thresholds)\ntest_kappa = quadratic_weighted_kappa(y_test, y_test_pred)\ntest_accuracy = accuracy_score(y_test, y_test_pred)\ntest_f1 = f1_score(y_test, y_test_pred, average='weighted')\nprint(f\"Test QWK: {Fore.GREEN}{Style.BRIGHT}{test_kappa:.4f}{Style.RESET_ALL}\")\nprint(f\"Test Accuracy: {Fore.GREEN}{Style.BRIGHT}{test_accuracy:.4f}{Style.RESET_ALL}\")\nprint(f\"Test F1: {Fore.GREEN}{Style.BRIGHT}{test_f1:.4f}{Style.RESET_ALL}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:08:49.767439Z","iopub.execute_input":"2024-12-11T00:08:49.767799Z","iopub.status.idle":"2024-12-11T00:08:49.806225Z","shell.execute_reply.started":"2024-12-11T00:08:49.767757Z","shell.execute_reply":"2024-12-11T00:08:49.805043Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"make_cm(y_test, y_test_pred, \"Voting Regressor - Test\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:08:49.807291Z","iopub.execute_input":"2024-12-11T00:08:49.807662Z","iopub.status.idle":"2024-12-11T00:08:50.081263Z","shell.execute_reply.started":"2024-12-11T00:08:49.807627Z","shell.execute_reply":"2024-12-11T00:08:50.080365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train = train_df.drop('sii', axis=1)\ny_train = train_df['sii'].astype(int)\ny_train_pred = best_model.predict(X_train)\ny_train_pred = apply_thresholds(y_train_pred, thresholds)\nmake_cm(y_train, y_train_pred, \"Voting Regressor - Train\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:08:50.082370Z","iopub.execute_input":"2024-12-11T00:08:50.082678Z","iopub.status.idle":"2024-12-11T00:08:50.398832Z","shell.execute_reply.started":"2024-12-11T00:08:50.082650Z","shell.execute_reply":"2024-12-11T00:08:50.397936Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model 2.5 - Stacking Regressor","metadata":{}},{"cell_type":"code","source":"# Create an ensemble using VotingClassifier with soft voting\nensemble_model = StackingRegressor(\n    estimators=[('lgb', lgb_model), \n                ('xgb', xgb_model), \n                ('cat', catb_model)\n                # ('rf', rf) \n                # ,('svc', svc)\n                ],\n    # weights=[4.0, 4.0, 5.0],\n    n_jobs=-1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:08:50.399946Z","iopub.execute_input":"2024-12-11T00:08:50.400221Z","iopub.status.idle":"2024-12-11T00:08:50.404439Z","shell.execute_reply.started":"2024-12-11T00:08:50.400194Z","shell.execute_reply":"2024-12-11T00:08:50.403481Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_model, thresholds = train_and_evaluate_regressor(ensemble_model, param_dist, n_iter=2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:08:50.405485Z","iopub.execute_input":"2024-12-11T00:08:50.405874Z","iopub.status.idle":"2024-12-11T00:09:23.259012Z","shell.execute_reply.started":"2024-12-11T00:08:50.405848Z","shell.execute_reply":"2024-12-11T00:09:23.258003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = test_df.drop('sii', axis=1)\ny_test = test_df['sii'].astype(int)\ny_test_pred = best_model.predict(X_test)\ny_test_pred = apply_thresholds(y_test_pred, thresholds)\ntest_kappa = quadratic_weighted_kappa(y_test, y_test_pred)\ntest_accuracy = accuracy_score(y_test, y_test_pred)\ntest_f1 = f1_score(y_test, y_test_pred, average='weighted')\nprint(f\"Test QWK: {Fore.GREEN}{Style.BRIGHT}{test_kappa:.4f}{Style.RESET_ALL}\")\nprint(f\"Test Accuracy: {Fore.GREEN}{Style.BRIGHT}{test_accuracy:.4f}{Style.RESET_ALL}\")\nprint(f\"Test F1: {Fore.GREEN}{Style.BRIGHT}{test_f1:.4f}{Style.RESET_ALL}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:09:23.260225Z","iopub.execute_input":"2024-12-11T00:09:23.260517Z","iopub.status.idle":"2024-12-11T00:09:23.299963Z","shell.execute_reply.started":"2024-12-11T00:09:23.260489Z","shell.execute_reply":"2024-12-11T00:09:23.298863Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model 3 - TabNet","metadata":{}},{"cell_type":"code","source":"\n# Function that instantiates a tabnet model.\ndef create_tabnet(n_d=32, n_steps=5, lr=0.02, gamma=1.5, \n                  n_independent=2, n_shared=2, lambda_sparse=1e-4, \n                  momentum=0.3, is_classifier=True):\n    if is_classifier:\n        tabnet_model = TabNetClassifier(\n            n_d=n_d,\n            n_a=n_d,\n            n_steps=n_steps,\n            gamma=gamma,\n            lambda_sparse=lambda_sparse,\n            seed=2024,\n            verbose=1,\n            device_name='cuda',\n            optimizer_fn=torch.optim.Adam,\n            scheduler_params = {\"gamma\": 0.95,\n                            \"step_size\": 20},\n            scheduler_fn=torch.optim.lr_scheduler.StepLR, epsilon=1e-15\n        )\n    else:\n        tabnet_model = TabNetRegressor(\n            n_d=n_d,\n            n_a=n_d,\n            n_steps=n_steps,\n            gamma=gamma,\n            lambda_sparse=lambda_sparse,\n            seed=2024,\n            verbose=1,\n            device_name='cuda',\n            optimizer_fn=torch.optim.Adam,\n            scheduler_params = {\"gamma\": 0.95,\n                            \"step_size\": 20},\n            scheduler_fn=torch.optim.lr_scheduler.StepLR, epsilon=1e-15\n        )\n    return tabnet_model\n                  \ndef getBestTabNetModel(is_classifier):\n    # Generate the parameter grid.\n    param_grid = dict(n_d = [64],\n                    n_a = [64],\n                    n_steps = [3],\n                    gamma = [1.3],\n                    lambda_sparse = [1e-2, 1e-3, 1e-4],\n                    momentum = [0.2],\n                    n_shared = [2],\n                    n_independent = [2],\n    )\n\n    grid = ParameterGrid(param_grid)\n\n    search_results = pd.DataFrame() \n    best_score = -np.inf\n    best_model = None\n    for params in tqdm(grid, desc=\"Hyperparameter Search\", ncols=100):\n        X_train, X_val, y_train, y_val = train_test_split(X.values.astype(np.float32), y.values.astype(np.int8), test_size=0.2, stratify=y)\n        if is_classifier == False:\n            y_train = y_train.reshape(-1, 1)\n            y_val = y_val.reshape(-1, 1)\n            eval_metric = ['rmse']\n        else:\n            eval_metric = ['logloss']\n        tqdm.write(f\"Before training - y_train shape: {str(y_train.shape)}, y_val shape: {str(y_val.shape)}\")\n        params['n_a'] = params['n_d'] # n_a=n_d always per the paper\n        tabnet = create_tabnet(is_classifier=is_classifier)\n        tabnet.set_params(**params)\n        tabnet.fit(X_train=X_train, y_train=y_train, eval_set=[(X_val, y_val)], eval_name=['validation'], eval_metric=eval_metric, max_epochs=200, patience=20,\n                batch_size=1024, virtual_batch_size=128)\n        if is_classifier:\n            y_pred = np.argmax(tabnet.predict_proba(X_val), axis=1)\n            score = quadratic_weighted_kappa(y_val, y_pred)  # Convert probabilities to class labels and calculate accuracy\n        else:\n            y_pred = tabnet.predict(X_val)\n            score = np.sqrt(mean_squared_error(y_val.astype(np.float32), y_pred))\n        if score > best_score:\n            best_model = tabnet\n            best_score = score\n    return best_model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:09:23.301302Z","iopub.execute_input":"2024-12-11T00:09:23.301772Z","iopub.status.idle":"2024-12-11T00:09:23.328779Z","shell.execute_reply.started":"2024-12-11T00:09:23.301721Z","shell.execute_reply":"2024-12-11T00:09:23.327631Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Classifier","metadata":{}},{"cell_type":"code","source":"best_model = getBestTabNetModel(is_classifier=True)\ntrain_X = X.values.astype(np.float32)\ntrain_pred = np.argmax(best_model.predict_proba(train_X), axis=1)\n\ntrain_kappa = quadratic_weighted_kappa(y, train_pred)\ntrain_accuracy = accuracy_score(y, train_pred)\ntrain_f1 = f1_score(y, train_pred, average='weighted')\n\nprint(f\"Best Train QWK: {train_kappa:.4f}\")\nprint(f\"Best Train Accuracy: {train_accuracy:.4f}\")\nprint(f\"Best Train F1: {train_f1:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:09:23.330028Z","iopub.execute_input":"2024-12-11T00:09:23.330347Z","iopub.status.idle":"2024-12-11T00:09:48.222002Z","shell.execute_reply.started":"2024-12-11T00:09:23.330312Z","shell.execute_reply":"2024-12-11T00:09:48.220825Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = test_df.drop('sii', axis=1)\ny_test = test_df['sii'].astype(int)\ntest_X = X_test.values.astype(np.float32)\ny_test_pred = np.argmax(best_model.predict_proba(test_X), axis=1)\ntest_kappa = quadratic_weighted_kappa(y_test, y_test_pred)\ntest_accuracy = accuracy_score(y_test, y_test_pred)\ntest_f1 = f1_score(y_test, y_test_pred, average='weighted')\n\nprint(f\"Best Test QWK: {test_kappa:.4f}\")\nprint(f\"Best Train Accuracy: {test_accuracy:.4f}\")\nprint(f\"Best Train F1: {test_f1:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:09:48.223488Z","iopub.execute_input":"2024-12-11T00:09:48.224202Z","iopub.status.idle":"2024-12-11T00:09:48.274957Z","shell.execute_reply.started":"2024-12-11T00:09:48.224141Z","shell.execute_reply":"2024-12-11T00:09:48.273809Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:09:48.276226Z","iopub.execute_input":"2024-12-11T00:09:48.276617Z","iopub.status.idle":"2024-12-11T00:09:48.283877Z","shell.execute_reply.started":"2024-12-11T00:09:48.276561Z","shell.execute_reply":"2024-12-11T00:09:48.282796Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Regressor","metadata":{}},{"cell_type":"code","source":"y.values.astype(np.int8).reshape(-1, 1).shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:09:48.285182Z","iopub.execute_input":"2024-12-11T00:09:48.285516Z","iopub.status.idle":"2024-12-11T00:09:48.295553Z","shell.execute_reply.started":"2024-12-11T00:09:48.285478Z","shell.execute_reply":"2024-12-11T00:09:48.294633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_model = getBestTabNetModel(is_classifier=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:09:48.297028Z","iopub.execute_input":"2024-12-11T00:09:48.297457Z","iopub.status.idle":"2024-12-11T00:10:19.635415Z","shell.execute_reply.started":"2024-12-11T00:09:48.297410Z","shell.execute_reply":"2024-12-11T00:10:19.634314Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"raw_preds = best_model.predict(X.values.astype(np.float32))\nthresholds = optimize_thresholds(y.values.astype(np.float32), raw_preds)\ntrain_pred = apply_thresholds(raw_preds, thresholds)\n\ntrain_kappa = quadratic_weighted_kappa(y, train_pred)\ntrain_accuracy = accuracy_score(y, train_pred)\ntrain_f1 = f1_score(y, train_pred, average='weighted')\n\nprint(f\"Best Train QWK: {train_kappa:.4f}\")\nprint(f\"Best Train Accuracy: {train_accuracy:.4f}\")\nprint(f\"Best Train F1: {train_f1:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:10:19.636880Z","iopub.execute_input":"2024-12-11T00:10:19.637244Z","iopub.status.idle":"2024-12-11T00:10:20.378575Z","shell.execute_reply.started":"2024-12-11T00:10:19.637205Z","shell.execute_reply":"2024-12-11T00:10:20.377629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = test_df.drop('sii', axis=1)\ny_test = test_df['sii'].astype(int)\ntest_X = X_test.values.astype(np.float32)\nraw_preds = best_model.predict(test_X)\ny_test_pred = apply_thresholds(raw_preds, thresholds)\ntest_kappa = quadratic_weighted_kappa(y_test, y_test_pred)\ntest_accuracy = accuracy_score(y_test, y_test_pred)\ntest_f1 = f1_score(y_test, y_test_pred, average='weighted')\n\nprint(f\"Best Test QWK: {test_kappa:.4f}\")\nprint(f\"Best Train Accuracy: {test_accuracy:.4f}\")\nprint(f\"Best Train F1: {test_f1:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T00:10:20.379923Z","iopub.execute_input":"2024-12-11T00:10:20.380683Z","iopub.status.idle":"2024-12-11T00:10:20.423042Z","shell.execute_reply.started":"2024-12-11T00:10:20.380637Z","shell.execute_reply":"2024-12-11T00:10:20.421819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}