{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30776,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.nn.utils.rnn import pad_sequence\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.preprocessing import QuantileTransformer, OrdinalEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.metrics import mean_squared_error, cohen_kappa_score\nimport os\nimport warnings\nfrom transformers import get_cosine_with_hard_restarts_schedule_with_warmup\nfrom scipy.optimize import minimize\nimport random\nfrom tqdm import tqdm\n\nWARMUP = True\nN_SPLITS = 5\nRANDOM_STATE = 1335\nEPOCHS = 8\nLR = 1e-3\nDATA_PATH = \"/kaggle/input/child-mind-institute-problematic-internet-use\"\nDROP_NAN = True\nCYCLES = 1\nQWK_WEIGHT = 0.75\nMSE_WEIGHT = 0.25\nWARMUP_RATIO = 0.0\nBATCH_SIZE = 32\nDROPOUT = 0.3\nPATIENCE = EPOCHS // 2\n\nwarnings.filterwarnings(\"ignore\")\nCOLS = [\n    \"Basic_Demos-Enroll_Season\",\n    \"CGAS-Season\",\n    \"Physical-Season\",\n    \"Fitness_Endurance-Season\",\n    \"FGC-Season\",\n    \"BIA-Season\",\n    \"PAQ_A-Season\",\n    \"PAQ_C-Season\",\n    \"SDS-Season\",\n    \"PreInt_EduHx-Season\",\n    \"FGC-FGC_PU\",\n    \"BIA-BIA_SMM\",\n    \"BIA-BIA_BMR\",\n    \"BIA-BIA_FFMI\",\n    \"BIA-BIA_TBW\",\n    \"Basic_Demos-Sex\",\n    \"BIA-BIA_LDM\",\n    \"Fitness_Endurance-Time_Mins\",\n    \"FGC-FGC_GSND\",\n    \"Basic_Demos-Age\",\n    \"FGC-FGC_GSD_Zone\",\n    \"BIA-BIA_Frame_num\",\n    \"Physical-HeartRate\",\n    \"FGC-FGC_SRL\",\n    \"Physical-Waist_Circumference\",\n    \"FGC-FGC_TL_Zone\",\n    \"Physical-Systolic_BP\",\n    \"CGAS-CGAS_Score\",\n    \"FGC-FGC_CU_Zone\",\n    \"BIA-BIA_ECW\",\n    \"FGC-FGC_SRR_Zone\",\n    \"PAQ_A-PAQ_A_Total\",\n    \"FGC-FGC_SRR\",\n    \"PreInt_EduHx-computerinternet_hoursday\",\n    \"SDS-SDS_Total_Raw\",\n    \"FGC-FGC_GSD\",\n    \"BIA-BIA_FFM\",\n    \"FGC-FGC_GSND_Zone\",\n    \"PAQ_C-PAQ_C_Total\",\n    \"BIA-BIA_BMI\",\n    \"FGC-FGC_PU_Zone\",\n    \"Fitness_Endurance-Time_Sec\",\n    \"Physical-Height\",\n    \"SDS-SDS_Total_T\",\n    \"FGC-FGC_CU\",\n    \"Physical-Weight\",\n    \"FGC-FGC_TL\",\n    \"Physical-Diastolic_BP\",\n    \"Physical-BMI\",\n    \"Fitness_Endurance-Max_Stage\",\n    \"FGC-FGC_SRL_Zone\",\n    \"BIA-BIA_FMI\",\n    \"BIA-BIA_BMC\",\n    \"BIA-BIA_DEE\",\n    \"BIA-BIA_ICW\",\n    \"BIA-BIA_Fat\",\n    \"BIA-BIA_LST\",\n    \"BIA-BIA_Activity_Level_num\",\n]\n\n\ndef set_seeds(seed):\n    random.seed(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n\nclass HybridDataset(Dataset):\n    def __init__(self, tabular_data, ids, targets=None, is_test=False):\n        self.categorical_data = torch.LongTensor(tabular_data[:, :10])\n        self.numerical_data = torch.FloatTensor(tabular_data[:, 10:])\n        self.ids = ids\n        self.is_test = is_test\n        if targets is not None:\n            self.targets = torch.LongTensor(targets)\n        else:\n            self.targets = None\n\n    def __len__(self):\n        return len(self.numerical_data)\n\n    def __getitem__(self, idx):\n        categorical = self.categorical_data[idx]\n        numerical = self.numerical_data[idx]\n\n        file_path = os.path.join(\n            DATA_PATH,\n            f\"series_{'test' if self.is_test else 'train'}.parquet/id={self.ids[idx]}/part-0.parquet\",\n        )\n        if os.path.exists(file_path):\n            time_series = pd.read_parquet(file_path)\n            numerical_ts = time_series.iloc[:, [1, 2, 3, 4, 5]].values\n            timestamp = (time_series.iloc[:, 9] / 3600000000000).round(2).values\n            weekday = time_series.iloc[:, 10].values - 1\n            assert np.all((weekday >= 0) & (weekday < 7)), \"Weekday values out of range\"\n            time_series = np.column_stack([numerical_ts, timestamp, weekday])\n        else:\n            time_series = np.zeros((1, 7))\n\n        time_series = torch.FloatTensor(time_series[::100])\n\n        if self.targets is not None:\n            target = self.targets[idx]\n            return categorical, numerical, time_series, target\n        return categorical, numerical, time_series\n\ndef collate_fn(batch):\n    categorical, numerical, time_series, targets = zip(*batch)\n    time_series_padded = pad_sequence(time_series, batch_first=True)\n    return (\n        torch.stack(categorical),\n        torch.stack(numerical),\n        time_series_padded,\n        torch.stack(targets),\n    )\n\ndef collate_fn_test(batch):\n    categorical, numerical, time_series = zip(*batch)\n    time_series_padded = pad_sequence(time_series, batch_first=True)\n    return (\n        torch.stack(categorical),\n        torch.stack(numerical),\n        time_series_padded,\n    )\n\ndef swish(x):\n    return x * torch.sigmoid(x)\n\nclass WaveBlock(nn.Module):\n    def __init__(self, in_channels, out_channels, dilation_rates, kernel_size):\n        super(WaveBlock, self).__init__()\n        self.num_rates = dilation_rates\n        self.convs = nn.ModuleList()\n        self.filter_convs = nn.ModuleList()\n        self.gate_convs = nn.ModuleList()\n\n        self.convs.append(nn.Conv1d(in_channels, out_channels, kernel_size=1))\n        dilation_rates = [2**i for i in range(dilation_rates)]\n        for dilation_rate in dilation_rates:\n            self.filter_convs.append(\n                nn.Conv1d(\n                    out_channels,\n                    out_channels,\n                    kernel_size=kernel_size,\n                    padding=int((dilation_rate * (kernel_size - 1)) / 2),\n                    dilation=dilation_rate,\n                )\n            )\n            self.gate_convs.append(\n                nn.Conv1d(\n                    out_channels,\n                    out_channels,\n                    kernel_size=kernel_size,\n                    padding=int((dilation_rate * (kernel_size - 1)) / 2),\n                    dilation=dilation_rate,\n                )\n            )\n            self.convs.append(nn.Conv1d(out_channels, out_channels, kernel_size=1))\n\n    def forward(self, x):\n        x = self.convs[0](x)\n        res = x\n        for i in range(self.num_rates):\n            x = swish(self.filter_convs[i](x)) * torch.sigmoid(self.gate_convs[i](x))\n            x = self.convs[i + 1](x)\n            res = 0.3 * res + 0.7 * x\n        return res\n\nclass TimeSeriesEncoder(nn.Module):\n    def __init__(self, input_dim, hidden_dim, n_layers=3, kernel_size=3):\n        super().__init__()\n        self.input_conv = nn.Conv1d(input_dim, hidden_dim, 1)\n        self.wave_block = WaveBlock(hidden_dim, hidden_dim, n_layers, kernel_size)\n        self.output_conv = nn.Conv1d(hidden_dim, hidden_dim, 1)\n        self.global_pooling = nn.AdaptiveAvgPool1d(1)\n\n        self.weekday_embedding = nn.Embedding(7, 8)\n\n        self.fc = nn.Linear(hidden_dim + 8, hidden_dim)\n        self.layer_norm = nn.LayerNorm(hidden_dim)\n        self.dropout = nn.Dropout(0.3)\n\n    def forward(self, x):\n        x_num, x_cat = x[:, :, :-1], x[:, :, -1].long()\n\n        x_num = x_num.transpose(1, 2)\n        x_num = self.input_conv(x_num)\n        x_num = self.wave_block(x_num)\n        x_num = self.output_conv(x_num)\n        x_num = self.global_pooling(x_num).squeeze(2)\n\n        x_cat = self.weekday_embedding(x_cat).mean(dim=1)\n\n        x_combined = torch.cat([x_num, x_cat], dim=1)\n        x_combined = self.fc(x_combined)\n        x_combined = self.layer_norm(x_combined)\n        return self.dropout(x_combined)\n\nclass HybridModel(nn.Module):\n    def __init__(\n        self,\n        categorical_dims,\n        numerical_dim,\n        time_series_dim,\n        embedding_dim,\n        hidden_dim,\n    ):\n        super().__init__()\n        self.embeddings = nn.ModuleList(\n            [\n                nn.Embedding(dim + 1, embedding_dim, padding_idx=dim)\n                for dim in categorical_dims\n            ]\n        )\n        self.numerical_encoder = nn.Sequential(\n            nn.BatchNorm1d(numerical_dim),\n            nn.Linear(numerical_dim, hidden_dim),\n            nn.GELU(),\n            nn.Dropout(0.3),\n            nn.Linear(hidden_dim, hidden_dim),\n            nn.GELU(),\n            nn.Dropout(0.3),\n        )\n        self.time_series_encoder = TimeSeriesEncoder(time_series_dim - 1, hidden_dim)\n        combined_dim_with_ts = len(categorical_dims) * embedding_dim + hidden_dim * 2\n        combined_dim_without_ts = len(categorical_dims) * embedding_dim + hidden_dim\n        self.regressor_with_ts = nn.Sequential(\n            nn.Linear(combined_dim_with_ts, hidden_dim),\n            nn.GELU(),\n            nn.Dropout(0.3),\n            nn.Linear(hidden_dim, hidden_dim),\n            nn.GELU(),\n            nn.Dropout(0.3),\n            nn.Linear(hidden_dim, 1),\n        )\n        self.regressor_without_ts = nn.Sequential(\n            nn.Linear(combined_dim_without_ts, hidden_dim),\n            nn.GELU(),\n            nn.Dropout(0.3),\n            nn.Linear(hidden_dim, hidden_dim),\n            nn.GELU(),\n            nn.Dropout(0.3),\n            nn.Linear(hidden_dim, 1),\n        )\n\n    def forward(self, categorical, numerical, time_series):\n        embedded = []\n        for i, emb in enumerate(self.embeddings):\n            clamped_indices = torch.clamp(categorical[:, i], 0, emb.num_embeddings - 1)\n            embedded.append(emb(clamped_indices))\n        embedded = torch.cat(embedded, dim=1)\n        numerical_features = self.numerical_encoder(numerical)\n        if time_series.size(1) > 1:\n            time_series_features = self.time_series_encoder(time_series)\n            combined = torch.cat(\n                [embedded, numerical_features, time_series_features], dim=1\n            )\n            output = self.regressor_with_ts(combined).squeeze(1)\n            output = torch.sigmoid(output) * 3  # Ensure output is between 0 and 3\n\n        else:\n            combined = torch.cat([embedded, numerical_features], dim=1)\n            output =  self.regressor_without_ts(combined).squeeze(1)\n            output = torch.sigmoid(output) * 3  # Ensure output is between 0 and 3\n        \n        if torch.isnan(output).any():\n            print(f\"NaN in model output. Embedded: {embedded}, Numerical: {numerical_features}, Time series: {time_series_features}\")\n            output = torch.where(torch.isnan(output), torch.zeros_like(output), output)\n        \n        return output\n\nclass QWKLoss(nn.Module):\n    def __init__(self, num_classes=4, y_min=0, y_max=3, epsilon=1e-10):\n        super().__init__()\n        self.num_classes = num_classes\n        self.y_min = y_min\n        self.y_max = y_max\n        self.epsilon = epsilon\n\n    def forward(self, pred, target):\n        pred = pred.clamp(self.y_min, self.y_max)\n        target = target.float().clamp(self.y_min, self.y_max)\n\n        # MSE component\n        mse = 0.5 * torch.mean((pred - target) ** 2)\n\n        # QWK approximation component\n        a = torch.mean(target)\n        b = torch.mean((target - a) ** 2)\n        qwk_approx = 0.5 * torch.mean((pred - a) ** 2 + b)\n\n        loss = mse / (qwk_approx + self.epsilon)\n        return loss\n\n    def backward(self, pred, target):\n        pred = pred.clamp(self.y_min, self.y_max)\n        target = target.float().clamp(self.y_min, self.y_max)\n\n        a = torch.mean(target)\n        b = torch.mean((target - a) ** 2)\n\n        f = 0.5 * torch.sum((pred - target) ** 2)\n        g = 0.5 * torch.sum((pred - a) ** 2 + b)\n\n        df = pred - target\n        dg = pred - a\n\n        grad = (df / g - f * dg / (g ** 2)) / len(target)\n        return grad\n\ndef calculate_qwk(predictions, targets):\n    predictions = np.clip(predictions, 0, 3).round()\n    targets = np.clip(targets, 0, 3).round()\n    return cohen_kappa_score(targets, predictions, weights=\"quadratic\")\n\ndef train_and_evaluate(\n    model, train_loader, val_loader, epochs, lr, device, patience=10, fold=1\n):\n    criterion_qwk = QWKLoss()\n    optimizer = optim.AdamW(model.parameters(), lr=lr, weight_decay=1e-4)\n\n    total_steps = len(train_loader) * epochs\n\n    scheduler = get_cosine_with_hard_restarts_schedule_with_warmup(\n        optimizer,\n        num_warmup_steps=int(0.1 * total_steps) if WARMUP else 0,\n        num_training_steps=total_steps,\n        num_cycles=1,\n    )\n\n    best_val_qwk = float(\"-inf\")\n    epochs_no_improve = 0\n    oof_predictions = np.zeros(len(val_loader.dataset))\n    oof_targets = np.zeros(len(val_loader.dataset))\n    \n    pbar = tqdm(range(epochs), desc=f'Fold {fold} Training')\n    \n    for epoch in pbar:\n        model.train()\n        train_loss = 0.0\n        for categorical, numerical, time_series, targets in train_loader:\n            categorical, numerical, time_series, targets = (\n                categorical.to(device),\n                numerical.to(device),\n                time_series.to(device),\n                targets.to(device),\n            )\n            optimizer.zero_grad()\n            outputs = model(categorical, numerical, time_series)\n            loss = criterion_qwk(outputs, targets)\n            \n            if torch.isnan(loss):\n                print(f\"NaN loss encountered. Outputs: {outputs}, Targets: {targets}\")\n                continue\n            \n            grad = criterion_qwk.backward(outputs, targets)\n            outputs.backward(gradient=grad)\n            \n            torch.nn.utils.clip_grad_norm_(model.parameters(), max_norm=1.0)\n            optimizer.step()\n            scheduler.step()\n            train_loss += loss.item()\n\n        model.eval()\n        val_loss = 0.0\n        val_preds = []\n        val_targets = []\n        with torch.no_grad():\n            for i, (categorical, numerical, time_series, targets) in enumerate(\n                val_loader\n            ):\n                categorical, numerical, time_series, targets = (\n                    categorical.to(device),\n                    numerical.to(device),\n                    time_series.to(device),\n                    targets.to(device),\n                )\n                outputs = model(categorical, numerical, time_series)\n                loss = criterion_qwk(outputs, targets)\n                \n                val_loss += loss.item()\n\n                val_preds.extend(outputs.cpu().numpy())\n                val_targets.extend(targets.cpu().numpy())\n\n        val_loss /= len(val_loader)\n        val_mse = mean_squared_error(val_targets, val_preds)\n        val_rmse = np.sqrt(val_mse)\n        val_qwk = calculate_qwk(val_preds, val_targets)\n\n        pbar.set_description(\n            f'Fold {fold} | Loss: {train_loss/len(train_loader):.4f} | Val Loss: {val_loss:.4f} | Val RMSE: {val_rmse:.4f} | Val QWK: {val_qwk:.4f}'\n        )\n\n        if val_qwk > best_val_qwk:\n            best_val_qwk = val_qwk\n            torch.save(model.state_dict(), f\"best_model_fold_{fold}.pth\")\n            epochs_no_improve = 0\n        else:\n            epochs_no_improve += 1\n\n        if epochs_no_improve == patience:\n            pbar.write(\"Early stopping triggered\")\n            break\n\n    model.load_state_dict(torch.load(f\"best_model_fold_{fold}.pth\"))\n    model.eval()\n    with torch.no_grad():\n        for i, (categorical, numerical, time_series, targets) in enumerate(val_loader):\n            categorical, numerical, time_series, targets = (\n                categorical.to(device),\n                numerical.to(device),\n                time_series.to(device),\n                targets.to(device),\n            )\n            outputs = model(categorical, numerical, time_series)\n            start_idx = i * val_loader.batch_size\n            end_idx = start_idx + targets.size(0)\n            oof_predictions[start_idx:end_idx] = outputs.cpu().numpy()\n            oof_targets[start_idx:end_idx] = targets.cpu().numpy()\n    fold_qwk = cohen_kappa_score(\n        oof_predictions.round(), oof_targets, weights=\"quadratic\"\n    )\n    print(f\"Fold {fold} QWK: {fold_qwk:.4f}\")\n\n    return fold_qwk, oof_predictions, oof_targets\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(\n        oof_non_rounded < thresholds[0],\n        0,\n        np.where(\n            oof_non_rounded < thresholds[1],\n            1,\n            np.where(oof_non_rounded < thresholds[2], 2, 3),\n        ),\n    )\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -cohen_kappa_score(y_true, rounded_p, weights=\"quadratic\")\n\ndef train():\n    train = pd.read_csv(os.path.join(DATA_PATH, \"train.csv\"))\n    if DROP_NAN:\n        train = train.dropna(subset=\"sii\")\n    else:\n        train[\"sii\"] = train[\"sii\"].fillna(train[\"sii\"].mean())\n    tabular_data = train[COLS]\n    targets = train[\"sii\"].values\n    ids = train[\"id\"].values\n\n    categorical_features = list(range(10))\n    numerical_features = list(range(10, 58))\n\n    preprocessor = ColumnTransformer(\n        transformers=[\n            (\n                \"num\",\n                Pipeline(\n                    [\n                        (\"imputer\", SimpleImputer(strategy=\"median\")),\n                        (\"scaler\", QuantileTransformer()),\n                    ]\n                ),\n                numerical_features,\n            ),\n            (\n                \"cat\",\n                Pipeline(\n                    [\n                        (\n                            \"imputer\",\n                            SimpleImputer(strategy=\"constant\", fill_value=\"missing\"),\n                        ),\n                        (\n                            \"encoder\",\n                            OrdinalEncoder(\n                                handle_unknown=\"use_encoded_value\", unknown_value=-1\n                            ),\n                        ),\n                    ]\n                ),\n                categorical_features,\n            ),\n        ]\n    )\n\n    tabular_data_processed = preprocessor.fit_transform(tabular_data)\n\n    ts_indicator = np.array(\n        [\n            (\n                1\n                if os.path.exists(\n                    os.path.join(\n                        DATA_PATH, f\"series_train.parquet/id={x}/part-0.parquet\"\n                    )\n                )\n                else 0\n            )\n            for x in ids\n        ]\n    ).reshape(-1, 1)\n    tabular_data_processed = np.column_stack([tabular_data_processed, ts_indicator])\n\n    categorical_dims = [\n        len(\n            preprocessor.named_transformers_[\"cat\"]\n            .named_steps[\"encoder\"]\n            .categories_[i]\n        )\n        for i in range(len(categorical_features))\n    ]\n\n    skf = StratifiedKFold(n_splits=N_SPLITS, shuffle=True, random_state=RANDOM_STATE)\n\n    fold_qwks = []\n    all_oof_predictions = np.zeros(len(targets))\n    all_oof_targets = np.zeros(len(targets))\n    for fold, (train_idx, val_idx) in enumerate(\n        skf.split(tabular_data_processed, targets), 1\n    ):\n        tabular_train, tabular_val = (\n            tabular_data_processed[train_idx],\n            tabular_data_processed[val_idx],\n        )\n        ids_train, ids_val = ids[train_idx], ids[val_idx]\n        y_train, y_val = targets[train_idx], targets[val_idx]\n\n        train_dataset = HybridDataset(tabular_train, ids_train, y_train)\n        val_dataset = HybridDataset(tabular_val, ids_val, y_val)\n        train_loader = DataLoader(\n            train_dataset,\n            batch_size=64,\n            shuffle=True,\n            collate_fn=collate_fn,\n            num_workers=4,\n        )\n        val_loader = DataLoader(\n            val_dataset,\n            batch_size=64,\n            shuffle=False,\n            collate_fn=collate_fn,\n            num_workers=4,\n        )\n\n        device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n        model = HybridModel(\n            categorical_dims=categorical_dims,\n            numerical_dim=tabular_train.shape[1] - len(categorical_dims),\n            time_series_dim=7,\n            embedding_dim=8,\n            hidden_dim=128,\n        ).to(device)\n\n        best_val_loss, oof_predictions, oof_targets = train_and_evaluate(\n            model,\n            train_loader,\n            val_loader,\n            epochs=EPOCHS,\n            lr=LR,\n            device=device,\n            patience=PATIENCE,\n            fold=fold,\n        )\n        fold_qwks.append(best_val_loss)\n        all_oof_predictions[val_idx] = oof_predictions\n        all_oof_targets[val_idx] = oof_targets\n\n    oof_mse = mean_squared_error(all_oof_targets, all_oof_predictions)\n    oof_rmse = np.sqrt(oof_mse)\n\n    Kappa_optimizer = minimize(\n        evaluate_predictions,\n        x0=[0.5, 1.5, 2.5],\n        args=(all_oof_targets, all_oof_predictions),\n        method=\"Nelder-Mead\",\n    )\n\n    assert Kappa_optimizer.success, \"Optimization did not converge.\"\n    np.save(\"oof_predictions.npy\", all_oof_predictions)\n    oof_tuned = threshold_Rounder(all_oof_predictions, Kappa_optimizer.x)\n    kappa = cohen_kappa_score(all_oof_targets, oof_tuned, weights=\"quadratic\")\n    oof_not_tuned = np.round(all_oof_predictions).astype(int)\n    oof_qwk = cohen_kappa_score(all_oof_targets, oof_not_tuned, weights=\"quadratic\")\n\n    print(f\"Overall OOF RMSE: {oof_rmse:.4f}\")\n    print(f\"Average OOF QWK: {np.mean(fold_qwks):.4f}\")\n    print(f\"Optimized QWK: {kappa:.4f}, not optimized QWK: {oof_qwk:.4f}\")\n    print(f\"Optimal thresholds: {Kappa_optimizer.x}\")\n\n\n    return model, preprocessor, fold_qwks, Kappa_optimizer.x\n\ndef inference(model, preprocessor, fold_qwks, optimal_thresholds, device=\"cuda\"):\n    test_predictions = []\n    test = pd.read_csv(os.path.join(DATA_PATH, \"test.csv\"))\n    tabular_data = test[COLS]\n    ids = test[\"id\"].values\n\n    tabular_data_processed = preprocessor.transform(tabular_data)\n\n    ts_indicator = np.array(\n        [\n            (\n                1\n                if os.path.exists(\n                    os.path.join(\n                        DATA_PATH, f\"series_test.parquet/id={x}/part-0.parquet\"\n                    )\n                )\n                else 0\n            )\n            for x in ids\n        ]\n    ).reshape(-1, 1)\n    tabular_data_processed = np.column_stack([tabular_data_processed, ts_indicator])\n\n    test_dataset = HybridDataset(tabular_data_processed, ids, is_test=True)\n    test_loader = DataLoader(\n        test_dataset,\n        batch_size=64,\n        shuffle=False,\n        collate_fn=collate_fn_test,\n        num_workers=4,\n    )\n\n    for fold in range(N_SPLITS):\n        model.load_state_dict(torch.load(f\"best_model_fold_{fold+1}.pth\"))\n        model.eval()\n        fold_preds = []\n        with torch.no_grad():\n            for categorical, numerical, time_series in test_loader:\n                categorical, numerical, time_series = (\n                    categorical.to(device),\n                    numerical.to(device),\n                    time_series.to(device),\n                )\n                outputs = model(categorical, numerical, time_series)\n                fold_preds.extend(outputs.cpu().numpy())\n\n        test_predictions.append(fold_preds)\n\n    print(\"\\nCross-validation results:\")\n    for fold, loss in enumerate(fold_qwks, 1):\n        print(f\"Fold {fold} QWK: {loss:.4f}\")\n    print(f\"Mean QWK: {np.mean(fold_qwks):.4f}\")\n\n    final_predictions = np.mean(test_predictions, axis=0)\n\n    final_predictions_tuned = threshold_Rounder(final_predictions, optimal_thresholds)\n\n    test[\"sii\"] = final_predictions_tuned\n    test[[\"id\", \"sii\"]].to_csv(\"submission.csv\", index=False)\n\nif __name__ == \"__main__\":\n    set_seeds(RANDOM_STATE)\n    model, preprocessor, fold_qwks, optimal_thresholds = train()\n    inference(model, preprocessor, fold_qwks, optimal_thresholds)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-25T15:19:46.787614Z","iopub.execute_input":"2024-11-25T15:19:46.787914Z"},"trusted":true},"outputs":[],"execution_count":null}]}