{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Phase 1: Foundation Models Linear Probing\nSince embeddings are already extracted and saved as `.npz` files, we only need to train a lightweight classification head on top of them.","metadata":{}},{"cell_type":"code","source":"import os\nimport glob\nimport numpy as np\nimport json\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, TensorDataset\nfrom sklearn.metrics import accuracy_score, cohen_kappa_score, f1_score\nfrom sklearn.model_selection import train_test_split\n\nprint(\"PyTorch Version:\", torch.__version__)\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(\"Using device:\", DEVICE)\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Configuration\nEPOCHS = 30\nBATCH_SIZE = 128\nLR = 1e-3\nWEIGHT_DECAY = 1e-4\nSEED = 42\n\ntorch.manual_seed(SEED)\nnp.random.seed(SEED)\n\nOUT_DIR = \"/kaggle/working/outputs\"\nos.makedirs(OUT_DIR, exist_ok=True)\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the Linear Probe Model\nclass LinearProbe(nn.Module):\n    def __init__(self, input_dim, num_classes=5):\n        super().__init__()\n        self.head = nn.Sequential(\n            nn.BatchNorm1d(input_dim),\n            nn.Dropout(0.3),\n            nn.Linear(input_dim, 512),\n            nn.ReLU(),\n            nn.BatchNorm1d(512),\n            nn.Dropout(0.3),\n            nn.Linear(512, num_classes)\n        )\n        \n    def forward(self, x):\n        return self.head(x)\n\ndef train_linear_probe(name, X, y):\n    print(f\"\\n{'='*50}\\nTraining Linear Probe for {name}\\n{'='*50}\")\n    print(f\"Input shape: {X.shape}, Labels shape: {y.shape}\")\n    \n    # Train / Val / Test Split (Stratified)\n    # Using a standard 70/15/15 split for the extracted embeddings\n    X_train_val, X_test, y_train_val, y_test = train_test_split(X, y, test_size=0.15, stratify=y, random_state=SEED)\n    X_train, X_val, y_train, y_val = train_test_split(X_train_val, y_train_val, test_size=0.176, stratify=y_train_val, random_state=SEED)\n    \n    train_ds = TensorDataset(torch.FloatTensor(X_train), torch.LongTensor(y_train))\n    val_ds = TensorDataset(torch.FloatTensor(X_val), torch.LongTensor(y_val))\n    test_ds = TensorDataset(torch.FloatTensor(X_test), torch.LongTensor(y_test))\n    \n    train_loader = DataLoader(train_ds, batch_size=BATCH_SIZE, shuffle=True)\n    val_loader = DataLoader(val_ds, batch_size=BATCH_SIZE, shuffle=False)\n    test_loader = DataLoader(test_ds, batch_size=BATCH_SIZE, shuffle=False)\n    \n    input_dim = X.shape[1]\n    model = LinearProbe(input_dim).to(DEVICE)\n    \n    # Ordinal-aware BCE / CrossEntropy\n    # Using standard CrossEntropy for simplicity, could also use ordinal BCE\n    criterion = nn.CrossEntropyLoss()\n    optimizer = optim.AdamW(model.parameters(), lr=LR, weight_decay=WEIGHT_DECAY)\n    scheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='max', factor=0.5, patience=3)\n    \n    best_qwk = -1.0\n    best_model_weights = None\n    \n    for epoch in range(EPOCHS):\n        model.train()\n        train_loss = 0\n        for batch_x, batch_y in train_loader:\n            batch_x, batch_y = batch_x.to(DEVICE), batch_y.to(DEVICE)\n            optimizer.zero_grad()\n            out = model(batch_x)\n            loss = criterion(out, batch_y)\n            loss.backward()\n            optimizer.step()\n            train_loss += loss.item()\n            \n        # Validation\n        model.eval()\n        val_preds, val_targs = [], []\n        with torch.no_grad():\n            for batch_x, batch_y in val_loader:\n                batch_x = batch_x.to(DEVICE)\n                out = model(batch_x)\n                preds = torch.argmax(out, dim=1).cpu().numpy()\n                val_preds.extend(preds)\n                val_targs.extend(batch_y.numpy())\n                \n        val_qwk = cohen_kappa_score(val_targs, val_preds, weights='quadratic')\n        scheduler.step(val_qwk)\n        \n        if val_qwk > best_qwk:\n            best_qwk = val_qwk\n            best_model_weights = model.state_dict().copy()\n            \n        if (epoch+1) % 5 == 0 or epoch == 0:\n            print(f\"Epoch {epoch+1:2d}/{EPOCHS} | Train Loss: {train_loss/len(train_loader):.4f} | Val QWK: {val_qwk:.4f}\")\n            \n    print(f\"\\n[INFO] Best Validation QWK: {best_qwk:.4f}\")\n    \n    # Test Evaluation\n    model.load_state_dict(best_model_weights)\n    model.eval()\n    test_preds, test_targs = [], []\n    with torch.no_grad():\n        for batch_x, batch_y in test_loader:\n            batch_x = batch_x.to(DEVICE)\n            out = model(batch_x)\n            preds = torch.argmax(out, dim=1).cpu().numpy()\n            test_preds.extend(preds)\n            test_targs.extend(batch_y.numpy())\n            \n    test_acc = accuracy_score(test_targs, test_preds)\n    test_qwk = cohen_kappa_score(test_targs, test_preds, weights='quadratic')\n    test_f1 = f1_score(test_targs, test_preds, average='macro')\n    \n    print(f\"[RESULT] {name} -> Test Acc: {test_acc:.4f} | Test QWK: {test_qwk:.4f} | Test F1: {test_f1:.4f}\")\n    \n    return {\n        \"accuracy\": round(float(test_acc), 4),\n        \"qwk\": round(float(test_qwk), 4),\n        \"f1_macro\": round(float(test_f1), 4)\n    }\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Locate and Load NPZ files\ndef find_npz_files():\n    # Kaggle input paths usually mount under /kaggle/input\n    # Search all attached datasets for npz files\n    search_paths = [\n        \"/kaggle/input/**/*.npz\",\n        \"*.npz\"\n    ]\n    \n    npz_files = []\n    for pattern in search_paths:\n        npz_files.extend(glob.glob(pattern, recursive=True))\n        \n    return list(set(npz_files))\n\nfound_files = find_npz_files()\nprint(f\"Found {len(found_files)} .npz files:\")\nfor f in found_files:\n    print(f\" - {f}\")\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Run Linear Probing on all found embeddings\nresults = {}\n\nfor fpath in found_files:\n    try:\n        data = np.load(fpath, allow_pickle=True)\n        # Assuming the npz file has arrays. Common keys are 'features', 'embeddings', 'X', 'y', 'labels'\n        keys = list(data.keys())\n        print(f\"\\nLoading {fpath}...\")\n        print(f\"Keys found: {keys}\")\n        \n        # Heuristic to find features and labels\n        x_key = next((k for k in keys if k in ['features', 'embeddings', 'X', 'x']), keys[0])\n        y_key = next((k for k in keys if k in ['labels', 'targets', 'y']), None)\n        \n        X = data[x_key]\n        if y_key is not None:\n            y = data[y_key]\n        else:\n            # If there's no label key, assume the second key is labels\n            if len(keys) > 1:\n                y = data[keys[1]]\n            else:\n                print(f\"Skipping {fpath}: Could not find labels in npz.\")\n                continue\n                \n        # Ensure correct shapes\n        if len(X.shape) > 2:\n            X = X.reshape(X.shape[0], -1)\n            \n        name = os.path.basename(fpath).replace('.npz', '')\n        \n        res = train_linear_probe(name, X, y)\n        results[name] = res\n        \n    except Exception as e:\n        print(f\"Error processing {fpath}: {e}\")\n\n# Save final results\nresults_file = os.path.join(OUT_DIR, \"foundation_model_results.json\")\nwith open(results_file, \"w\") as f:\n    json.dump(results, f, indent=4)\n    \nprint(\"\\nAll done! Results saved to\", results_file)\nprint(json.dumps(results, indent=2))\n","metadata":{},"outputs":[],"execution_count":null}]}