{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isGpuEnabled":false,"isInternetEnabled":false,"language":"python","sourceType":"notebook"},"papermill":{"default_parameters":{},"duration":null,"end_time":null,"environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-06-24T08:02:56.413869+00:00","version":"2.7.0"},"colab":{"provenance":[]}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Environment Setup & Dependencies","metadata":{"papermill":{"duration":0.004134,"end_time":"2026-06-24T08:02:59.000055+00:00","exception":false,"start_time":"2026-06-24T08:02:58.995921+00:00","status":"completed"},"tags":[],"id":"ed486b8a"}},{"cell_type":"code","source":"!pip install -q segmentation-models-pytorch lightning albumentations scikit-image \\\n                streamlit onnxruntime fastapi uvicorn python-multipart huggingface_hub torchmetrics\n\nimport os\nimport cv2\nimport random\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import classification_report, confusion_matrix\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader\n\nimport lightning as L\nfrom lightning.pytorch.callbacks import ModelCheckpoint, EarlyStopping\nimport segmentation_models_pytorch as smp\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nfrom skimage.measure import label, regionprops","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-03T11:56:58.397748Z","iopub.execute_input":"2026-07-03T11:56:58.397980Z","iopub.status.idle":"2026-07-03T11:57:37.464474Z","shell.execute_reply.started":"2026-07-03T11:56:58.397915Z","shell.execute_reply":"2026-07-03T11:57:37.463432Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def seed_everything(seed=42):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n\nseed_everything(42)\n\nDATA_DIR = \"/kaggle/input/competitions/human-protein-atlas-image-classification\"\nTRAIN_CSV = os.path.join(DATA_DIR, \"train.csv\")\nTRAIN_IMG_DIR = os.path.join(DATA_DIR, \"train\")\nOUTPUT_DIR = \"/kaggle/working\"\nCACHE_DIR = \"/kaggle/working/hpa_cache\"\n\nprint(f\"CUDA Available: {torch.cuda.is_available()}\")\nif torch.cuda.is_available():\n    print(f\"Device Name: {torch.cuda.get_device_name(0)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-03T11:57:50.737771Z","iopub.execute_input":"2026-07-03T11:57:50.738140Z","iopub.status.idle":"2026-07-03T11:57:51.080422Z","shell.execute_reply.started":"2026-07-03T11:57:50.738095Z","shell.execute_reply":"2026-07-03T11:57:51.079564Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Advanced Exploratory Data Analysis (EDA)","metadata":{"papermill":{"duration":0.003624,"end_time":"2026-06-24T08:03:35.697841+00:00","exception":false,"start_time":"2026-06-24T08:03:35.694217+00:00","status":"completed"},"tags":[],"id":"cbf1926c"}},{"cell_type":"code","source":"df_metadata = pd.read_csv(TRAIN_CSV)\nprint(f\"📊 Total samples in metadata layout: {len(df_metadata)}\")\n\nfig, axes = plt.subplots(1, 2, figsize=(20, 6))\n\n# Flatten labels to calculate absolute target distributions\nall_targets = []\ndf_metadata['Target'].dropna().apply(lambda x: all_targets.extend([int(t) for t in x.split()]))\n\nsns.countplot(ax=axes[0], x=all_targets, palette=\"viridis\")\naxes[0].set_title(\"Distribution of Protein Cellular Localizations (Absolute Counts)\")\naxes[0].set_xlabel(\"Target Label ID\")\naxes[0].set_ylabel(\"Count\")\naxes[0].grid(axis='y', linestyle='--', alpha=0.5)\n\n# Label Cardinality: How many labels are assigned simultaneously per image?\nlabels_per_image = df_metadata['Target'].dropna().apply(lambda x: len(x.split()))\nsns.countplot(ax=axes[1], x=labels_per_image, palette=\"magma\")\naxes[1].set_title(\"Number of Target Labels Assigned per Image Instance\")\naxes[1].set_xlabel(\"Number of Simultaneous Targets\")\naxes[1].set_ylabel(\"Number of Images\")\naxes[1].grid(axis='y', linestyle='--', alpha=0.5)\nplt.tight_layout()\nplt.show()\n\n# Multi-spectral channel rendering verification\nsample_row = df_metadata.iloc[0]\nsample_id = sample_row['Id']\ntry:\n    b_img = cv2.imread(os.path.join(TRAIN_IMG_DIR, f\"{sample_id}_blue.png\"), cv2.IMREAD_GRAYSCALE)\n    r_img = cv2.imread(os.path.join(TRAIN_IMG_DIR, f\"{sample_id}_red.png\"), cv2.IMREAD_GRAYSCALE)\n    y_img = cv2.imread(os.path.join(TRAIN_IMG_DIR, f\"{sample_id}_yellow.png\"), cv2.IMREAD_GRAYSCALE)\n    g_img = cv2.imread(os.path.join(TRAIN_IMG_DIR, f\"{sample_id}_green.png\"), cv2.IMREAD_GRAYSCALE)\nexcept Exception:\n    b_img = r_img = y_img = g_img = np.random.randint(0, 255, (512, 512), dtype=np.uint8)\n\nfig, axes = plt.subplots(1, 4, figsize=(20, 5))\naxes[0].imshow(b_img, cmap='Blues'); axes[0].set_title(\"Blue: Nuclei\")\naxes[1].imshow(r_img, cmap='Reds'); axes[1].set_title(\"Red: Microtubules\")\naxes[2].imshow(y_img, cmap='Wistia'); axes[2].set_title(\"Yellow: Endoplasmic Reticulum\")\naxes[3].imshow(g_img, cmap='Greens'); axes[3].set_title(\"Green: Target Protein\")\nfor ax in axes:\n    ax.axis('off')\nplt.suptitle(f\"Multi-spectral Independent Channels for Sample: {sample_id}\", fontsize=14, fontweight='bold')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2026-07-03T11:57:54.846427Z","iopub.execute_input":"2026-07-03T11:57:54.847057Z","iopub.status.idle":"2026-07-03T11:57:56.569512Z","shell.execute_reply.started":"2026-07-03T11:57:54.847022Z","shell.execute_reply":"2026-07-03T11:57:56.568335Z"},"papermill":{"duration":1.250325,"end_time":"2026-06-24T08:03:36.951920+00:00","exception":false,"start_time":"2026-06-24T08:03:35.701595+00:00","status":"completed"},"tags":[],"id":"f350664b","outputId":"654a5143-fde3-4d37-dc3b-5d408b71ef2a","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Pipeline & Adaptive Channel Engineering","metadata":{"papermill":{"duration":0.013482,"end_time":"2026-06-24T08:03:36.979184+00:00","exception":false,"start_time":"2026-06-24T08:03:36.965702+00:00","status":"completed"},"tags":[],"id":"e0fb183a"}},{"cell_type":"code","source":"def apply_clahe_per_channel(img_fused, clip_limit=2.0, grid_size=(8, 8)):\n    clahe = cv2.createCLAHE(clipLimit=clip_limit, tileGridSize=grid_size)\n    equalized = np.zeros_like(img_fused)\n    for c in range(img_fused.shape[2]):\n        ch = img_fused[:, :, c]\n        if ch.dtype != np.uint8:\n            ch = cv2.normalize(ch, None, 0, 255, cv2.NORM_MINMAX).astype(np.uint8)\n        equalized[:, :, c] = clahe.apply(ch)\n    return equalized\n\ndef min_max_normalize(img_fused):\n    img_fused = img_fused.astype(np.float32)\n    for c in range(img_fused.shape[2]):\n        min_val = img_fused[:, :, c].min()\n        max_val = img_fused[:, :, c].max()\n        if (max_val - min_val) > 0:\n            img_fused[:, :, c] = (img_fused[:, :, c] - min_val) / (max_val - min_val)\n        else:\n            img_fused[:, :, c] = 0.0\n    return img_fused\n\nclass KaggleHPAMicroscopeDataset(Dataset):\n    def __init__(self, df, img_dir, transforms=None, cache_dir=None):\n        self.df = df.reset_index(drop=True)\n        self.img_dir = img_dir\n        self.transforms = transforms\n        self.cache_dir = cache_dir  # if set, loads from cache instead of processing live\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img_id = row['Id']\n\n        # Fast path: load from cache\n        if self.cache_dir:\n            cache_path = os.path.join(self.cache_dir, f\"{img_id}.npz\")\n            try:\n                data = np.load(cache_path)\n                fused = data['image']   # already float32, CLAHE'd, normalized\n                mask  = data['mask']    # already uint8 with classes 0/1/2\n            except Exception:\n                fused = np.zeros((512, 512, 4), dtype=np.float32)\n                mask  = np.zeros((512, 512), dtype=np.uint8)\n\n        # Slow path fallback (if no cache)\n        else:\n            try:\n                blue   = cv2.imread(os.path.join(self.img_dir, f\"{img_id}_blue.png\"),   cv2.IMREAD_GRAYSCALE)\n                red    = cv2.imread(os.path.join(self.img_dir, f\"{img_id}_red.png\"),    cv2.IMREAD_GRAYSCALE)\n                yellow = cv2.imread(os.path.join(self.img_dir, f\"{img_id}_yellow.png\"), cv2.IMREAD_GRAYSCALE)\n                green  = cv2.imread(os.path.join(self.img_dir, f\"{img_id}_green.png\"),  cv2.IMREAD_GRAYSCALE)\n            except Exception:\n                blue = None\n            if blue is None:\n                blue = red = yellow = green = np.zeros((512, 512), dtype=np.uint8)\n            fused = np.stack([blue, red, yellow, green], axis=-1)\n            fused = apply_clahe_per_channel(fused)\n            fused = min_max_normalize(fused)\n            _, binary_nuclei = cv2.threshold((fused[:, :, 0] * 255).astype(np.uint8), 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)\n            _, binary_cells  = cv2.threshold((fused[:, :, 1] * 255).astype(np.uint8), 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)\n            mask = np.zeros(fused.shape[:2], dtype=np.uint8)\n            mask[binary_cells > 0] = 1\n            mask[binary_nuclei > 0] = 2\n\n        if self.transforms:\n            augmented = self.transforms(image=fused, mask=mask)\n            return augmented['image'], augmented['mask'].long()\n\n        image_tensor = torch.tensor(fused, dtype=torch.float32).permute(2, 0, 1)\n        mask_tensor  = torch.tensor(mask, dtype=torch.long)\n        return image_tensor, mask_tensor","metadata":{"execution":{"iopub.status.busy":"2026-07-03T11:58:02.781800Z","iopub.execute_input":"2026-07-03T11:58:02.782807Z","iopub.status.idle":"2026-07-03T11:58:02.803095Z","shell.execute_reply.started":"2026-07-03T11:58:02.782753Z","shell.execute_reply":"2026-07-03T11:58:02.802105Z"},"papermill":{"duration":0.027929,"end_time":"2026-06-24T08:03:37.020570+00:00","exception":false,"start_time":"2026-06-24T08:03:36.992641+00:00","status":"completed"},"tags":[],"id":"26f73e90","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Subset Stratification","metadata":{"papermill":{"duration":0.013355,"end_time":"2026-06-24T08:03:37.048140+00:00","exception":false,"start_time":"2026-06-24T08:03:37.034785+00:00","status":"completed"},"tags":[],"id":"aad2575c"}},{"cell_type":"code","source":"df_metadata['cell_line_group'] = df_metadata['Target'].apply(lambda x: str(x).split()[0] if pd.notnull(x) else \"0\")\n\n# StratifiedKFold logic safely isolates validation blocks over the full distribution\nskf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\ndf_metadata['fold'] = -1\nfor fold, (train_idx, val_idx) in enumerate(skf.split(df_metadata, df_metadata['cell_line_group'])):\n    df_metadata.loc[val_idx, 'fold'] = fold\n\n# 🚀 Unlocking the full dataset mapping. All raw sub-sampling constraints removed.\ntrain_df = df_metadata[df_metadata['fold'] != 0].reset_index(drop=True)\nvalid_df = df_metadata[df_metadata['fold'] == 0].reset_index(drop=True)\n\n# Stratified subset — representative sample for training demo\n# Scale back up when more compute is available\nTRAIN_SUBSET = 6000\nVAL_SUBSET = 1500\ntrain_df = train_df.sample(n=min(TRAIN_SUBSET, len(train_df)), random_state=42).reset_index(drop=True)\nvalid_df = valid_df.sample(n=min(VAL_SUBSET, len(valid_df)), random_state=42).reset_index(drop=True)\n\nprint(f\"🚀 Training on FULL cross-validation layout: {len(train_df)} images\")\nprint(f\"📊 Validating on FULL cross-validation layout: {len(valid_df)} images\")","metadata":{"execution":{"iopub.status.busy":"2026-07-03T11:58:07.212277Z","iopub.execute_input":"2026-07-03T11:58:07.213037Z","iopub.status.idle":"2026-07-03T11:58:07.304967Z","shell.execute_reply.started":"2026-07-03T11:58:07.213003Z","shell.execute_reply":"2026-07-03T11:58:07.304123Z"},"papermill":{"duration":0.078266,"end_time":"2026-06-24T08:03:37.139670+00:00","exception":false,"start_time":"2026-06-24T08:03:37.061404+00:00","status":"completed"},"tags":[],"id":"44f9ebc5","outputId":"222c9467-26ec-495b-aeed-1b5ea4a06423","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Preprocess Cache","metadata":{"id":"xYCQEPPvlB18"}},{"cell_type":"code","source":"# Runs CLAHE + normalization + mask generation once per image and saves to .npz\n# __getitem__ does np.load() (free.99)\n\nimport os\n\nCACHE_DIR = \"/kaggle/working/hpa_cache\"\nos.makedirs(CACHE_DIR, exist_ok=True)\n\ndef preprocess_and_cache_dataset(df, img_dir, cache_dir):\n    skipped, processed = 0, 0\n    for idx, row in df.iterrows():\n        img_id = row['Id']\n        cache_path = os.path.join(cache_dir, f\"{img_id}.npz\")\n        if os.path.exists(cache_path):\n            skipped += 1\n            continue\n        try:\n            blue   = cv2.imread(os.path.join(img_dir, f\"{img_id}_blue.png\"),   cv2.IMREAD_GRAYSCALE)\n            red    = cv2.imread(os.path.join(img_dir, f\"{img_id}_red.png\"),    cv2.IMREAD_GRAYSCALE)\n            yellow = cv2.imread(os.path.join(img_dir, f\"{img_id}_yellow.png\"), cv2.IMREAD_GRAYSCALE)\n            green  = cv2.imread(os.path.join(img_dir, f\"{img_id}_green.png\"),  cv2.IMREAD_GRAYSCALE)\n            if blue is None:\n                raise ValueError(\"Missing channel\")\n        except Exception:\n            blue = red = yellow = green = np.zeros((512, 512), dtype=np.uint8)\n\n        fused = np.stack([blue, red, yellow, green], axis=-1)\n        fused = apply_clahe_per_channel(fused)\n        fused = min_max_normalize(fused)\n\n        _, binary_nuclei = cv2.threshold((fused[:, :, 0] * 255).astype(np.uint8), 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)\n        _, binary_cells  = cv2.threshold((fused[:, :, 1] * 255).astype(np.uint8), 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)\n        mask = np.zeros(fused.shape[:2], dtype=np.uint8)\n        mask[binary_cells > 0] = 1\n        mask[binary_nuclei > 0] = 2\n\n        np.savez_compressed(cache_path, image=fused.astype(np.float32), mask=mask)\n        processed += 1\n\n        if processed % 500 == 0:\n            print(f\"  Cached {processed} images...\")\n\n    print(f\"✅ Cache complete. Processed: {processed} | Already cached: {skipped}\")\n\nprint(\"⚙️  Building preprocessing cache (runs once, skips existing)...\")\n# Cache only what we'll actually train/validate on\npreprocess_and_cache_dataset(train_df, TRAIN_IMG_DIR, CACHE_DIR)\npreprocess_and_cache_dataset(valid_df, TRAIN_IMG_DIR, CACHE_DIR)","metadata":{"id":"ea9Bd2sKlFP0","trusted":true,"execution":{"iopub.status.busy":"2026-06-28T01:25:51.341846Z","iopub.execute_input":"2026-06-28T01:25:51.342666Z","iopub.status.idle":"2026-06-28T01:49:47.229566Z","shell.execute_reply.started":"2026-06-28T01:25:51.342633Z","shell.execute_reply":"2026-06-28T01:49:47.228833Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Check Cache","metadata":{}},{"cell_type":"code","source":"print(len(os.listdir(\"/kaggle/working/hpa_cache\")))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-03T11:58:12.082832Z","iopub.execute_input":"2026-07-03T11:58:12.083659Z","iopub.status.idle":"2026-07-03T11:58:12.096120Z","shell.execute_reply.started":"2026-07-03T11:58:12.083625Z","shell.execute_reply":"2026-07-03T11:58:12.095303Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Pretrained Multi-spectral UNet Adapter","metadata":{"papermill":{"duration":0.014412,"end_time":"2026-06-24T08:03:37.168190+00:00","exception":false,"start_time":"2026-06-24T08:03:37.153778+00:00","status":"completed"},"tags":[],"id":"ad780d9f"}},{"cell_type":"code","source":"class Pretrained4ChannelUNet(nn.Module):\n    def __init__(self, encoder_name=\"resnet34\", num_classes=3):\n        super().__init__()\n        self.backbone = smp.Unet(\n            encoder_name=encoder_name,\n            encoder_weights=\"imagenet\",\n            in_channels=3,\n            classes=num_classes\n        )\n        self._adapt_first_layer()\n\n    def _adapt_first_layer(self):\n        original_layer = self.backbone.encoder.conv1\n        weight_tensor = original_layer.weight.data\n        mean_weight_vector = torch.mean(weight_tensor, dim=1, keepdim=True)\n        new_4ch_weight_tensor = torch.cat([weight_tensor, mean_weight_vector], dim=1)\n\n        mutated_conv = nn.Conv2d(\n            in_channels=4,\n            out_channels=original_layer.out_channels,\n            kernel_size=original_layer.kernel_size,\n            stride=original_layer.stride,\n            padding=original_layer.padding,\n            bias=original_layer.bias is not None\n        )\n        mutated_conv.weight.data = new_4ch_weight_tensor\n        if original_layer.bias is not None:\n            mutated_conv.bias.data = original_layer.bias.data\n        self.backbone.encoder.conv1 = mutated_conv\n\n    def forward(self, x):\n        return self.backbone(x)","metadata":{"execution":{"iopub.status.busy":"2026-07-03T11:58:15.724579Z","iopub.execute_input":"2026-07-03T11:58:15.725705Z","iopub.status.idle":"2026-07-03T11:58:15.732786Z","shell.execute_reply.started":"2026-07-03T11:58:15.725657Z","shell.execute_reply":"2026-07-03T11:58:15.732175Z"},"papermill":{"duration":0.023377,"end_time":"2026-06-24T08:03:37.205462+00:00","exception":false,"start_time":"2026-06-24T08:03:37.182085+00:00","status":"completed"},"tags":[],"id":"da14500e","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Pipeline Component Testing (Sanity Verification)","metadata":{"papermill":{"duration":0.013759,"end_time":"2026-06-24T08:03:37.234389+00:00","exception":false,"start_time":"2026-06-24T08:03:37.220630+00:00","status":"completed"},"tags":[],"id":"62f6f184"}},{"cell_type":"code","source":"def test_pipeline_components():\n    print(\"🧪 Executing isolated unit health checks...\")\n    test_ds = KaggleHPAMicroscopeDataset(train_df.head(2), TRAIN_IMG_DIR)\n    img, msk = test_ds[0]\n    assert img.shape[0] == 4, f\"Expected 4 channels, got {img.shape}\"\n    print(f\"✅ Input shape: {img.shape} | Mask shape: {msk.shape}\")\n    print(\"✅ System structural integrity verified.\")\n\ntest_pipeline_components()","metadata":{"execution":{"iopub.status.busy":"2026-07-03T11:58:19.328530Z","iopub.execute_input":"2026-07-03T11:58:19.329128Z","iopub.status.idle":"2026-07-03T11:58:19.683209Z","shell.execute_reply.started":"2026-07-03T11:58:19.329093Z","shell.execute_reply":"2026-07-03T11:58:19.682392Z"},"papermill":{"duration":0.075268,"end_time":"2026-06-24T08:03:37.323544+00:00","exception":false,"start_time":"2026-06-24T08:03:37.248276+00:00","status":"completed"},"tags":[],"id":"6aa7e8e8","outputId":"95879aa8-a060-41e4-fe26-694a16b6e5bb","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Hybrid Structural Loss & Lightning Module Engine","metadata":{"papermill":{"duration":0.01322,"end_time":"2026-06-24T08:03:37.350987+00:00","exception":false,"start_time":"2026-06-24T08:03:37.337767+00:00","status":"completed"},"tags":[],"id":"605b8bb3"}},{"cell_type":"code","source":"import torchmetrics\n\nclass BiologicalHybridLoss(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.ce = nn.CrossEntropyLoss()\n\n    def forward(self, logits, targets):\n        return self.ce(logits, targets)\n\nclass MicroscopeSegmentationEngine(L.LightningModule):\n    def __init__(self):\n        super().__init__()\n        self.net = Pretrained4ChannelUNet()\n        self.loss_fn = BiologicalHybridLoss()\n        # Real pixel accuracy — 3 classes, ignores nothing\n        self.train_acc = torchmetrics.Accuracy(task=\"multiclass\", num_classes=3, average=\"macro\")\n        self.val_acc   = torchmetrics.Accuracy(task=\"multiclass\", num_classes=3, average=\"macro\")\n\n    def forward(self, x):\n        return self.net(x)\n\n    def training_step(self, batch, batch_idx):\n        x, y = batch\n        logits = self(x)\n        loss = self.loss_fn(logits, y)\n        preds = torch.argmax(logits, dim=1)\n        self.train_acc(preds, y)\n        self.log(\"train_loss\", loss, on_epoch=True, prog_bar=True)\n        self.log(\"train_acc\",  self.train_acc, on_epoch=True, prog_bar=True)\n        return loss\n\n    def validation_step(self, batch, batch_idx):\n        x, y = batch\n        logits = self(x)\n        loss = self.loss_fn(logits, y)\n        preds = torch.argmax(logits, dim=1)\n        self.val_acc(preds, y)\n        self.log(\"val_loss\", loss, on_epoch=True, prog_bar=True)\n        self.log(\"val_acc\",  self.val_acc, on_epoch=True, prog_bar=True)\n\n    def configure_optimizers(self):\n        return torch.optim.AdamW(self.parameters(), lr=2e-4, weight_decay=1e-4)","metadata":{"execution":{"iopub.status.busy":"2026-07-03T11:58:28.846514Z","iopub.execute_input":"2026-07-03T11:58:28.847264Z","iopub.status.idle":"2026-07-03T11:58:28.856443Z","shell.execute_reply.started":"2026-07-03T11:58:28.847231Z","shell.execute_reply":"2026-07-03T11:58:28.855738Z"},"papermill":{"duration":0.022523,"end_time":"2026-06-24T08:03:37.386866+00:00","exception":false,"start_time":"2026-06-24T08:03:37.364343+00:00","status":"completed"},"tags":[],"id":"97266f66","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Training ","metadata":{"papermill":{"duration":0.012954,"end_time":"2026-06-24T08:03:37.413838+00:00","exception":false,"start_time":"2026-06-24T08:03:37.400884+00:00","status":"completed"},"tags":[],"id":"074ad7ca"}},{"cell_type":"code","source":"from lightning.pytorch.loggers import CSVLogger\n\ncsv_logger = CSVLogger(save_dir=OUTPUT_DIR, name=\"lightning_logs\")\n\n\ndef get_train_augmentations():\n    return A.Compose([\n        # KEEP AT ORIGINAL RESOLUTION TO PROTECT MORPHOLOGICAL EDGES\n        A.Resize(512, 512), \n        A.HorizontalFlip(p=0.5),\n        A.VerticalFlip(p=0.5),\n        A.RandomRotate90(p=0.5),\n        A.RandomBrightnessContrast(brightness_limit=0.2, contrast_limit=0.2, p=0.4),\n        A.GaussianBlur(blur_limit=(3, 5), p=0.3),\n        ToTensorV2()\n    ], additional_targets={'mask': 'mask'})\n\ndef get_valid_augmentations():\n    return A.Compose([\n        A.Resize(512, 512),\n        ToTensorV2()\n    ], additional_targets={'mask': 'mask'})\n\ntrain_ds = KaggleHPAMicroscopeDataset(train_df, TRAIN_IMG_DIR, transforms=get_train_augmentations(), cache_dir=CACHE_DIR)\nvalid_ds = KaggleHPAMicroscopeDataset(valid_df, TRAIN_IMG_DIR, transforms=get_valid_augmentations(), cache_dir=CACHE_DIR)\n# 2. Maximize data throughput with Batch Size 32 to feed the GPU fully\ntrain_loader = DataLoader(train_ds, batch_size=32, shuffle=True,  num_workers=2, pin_memory=True, persistent_workers=True)\nvalid_loader = DataLoader(valid_ds, batch_size=32, shuffle=False, num_workers=2, pin_memory=True, persistent_workers=True)\nckpt_cb = ModelCheckpoint(monitor=\"val_loss\", dirpath=OUTPUT_DIR, filename=\"best_unet_model\", save_top_k=1, mode=\"min\")\nearly_cb = EarlyStopping(monitor=\"val_loss\", patience=1, mode=\"min\")\n\ntrainer = L.Trainer(\n    max_epochs=15,\n    accelerator=\"auto\",\n    devices=1,\n    precision=\"16-mixed\",\n    callbacks=[ckpt_cb, early_cb],\n    logger=csv_logger,\n    enable_progress_bar=False,\n    log_every_n_steps=10        \n)\n# Run the lightning fast training loop\ntrainer.fit(model=MicroscopeSegmentationEngine(), train_dataloaders=train_loader, val_dataloaders=valid_loader)\nprint(\"🎯 Emergency Fast Pipeline Finished Successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-02T22:36:00.342081Z","iopub.execute_input":"2026-07-02T22:36:00.343055Z","iopub.status.idle":"2026-07-02T23:10:48.669937Z","shell.execute_reply.started":"2026-07-02T22:36:00.343017Z","shell.execute_reply":"2026-07-02T23:10:48.668900Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from lightning.pytorch.loggers import CSVLogger\n\ncsv_logger = CSVLogger(save_dir=OUTPUT_DIR, name=\"lightning_logs\")\n\n\ndef get_train_augmentations():\n    return A.Compose([\n        # KEEP AT ORIGINAL RESOLUTION TO PROTECT MORPHOLOGICAL EDGES\n        A.Resize(512, 512), \n        A.HorizontalFlip(p=0.5),\n        A.VerticalFlip(p=0.5),\n        A.RandomRotate90(p=0.5),\n        A.RandomBrightnessContrast(brightness_limit=0.2, contrast_limit=0.2, p=0.4),\n        A.GaussianBlur(blur_limit=(3, 5), p=0.3),\n        ToTensorV2()\n    ], additional_targets={'mask': 'mask'})\n\ndef get_valid_augmentations():\n    return A.Compose([\n        A.Resize(512, 512),\n        ToTensorV2()\n    ], additional_targets={'mask': 'mask'})\n\ntrain_ds = KaggleHPAMicroscopeDataset(train_df, TRAIN_IMG_DIR, transforms=get_train_augmentations(), cache_dir=CACHE_DIR)\nvalid_ds = KaggleHPAMicroscopeDataset(valid_df, TRAIN_IMG_DIR, transforms=get_valid_augmentations(), cache_dir=CACHE_DIR)\n# 2. Maximize data throughput with Batch Size 32 to feed the GPU fully\ntrain_loader = DataLoader(train_ds, batch_size=32, shuffle=True,  num_workers=2, pin_memory=True, persistent_workers=True)\nvalid_loader = DataLoader(valid_ds, batch_size=32, shuffle=False, num_workers=2, pin_memory=True, persistent_workers=True)\nckpt_cb = ModelCheckpoint(monitor=\"val_loss\", dirpath=OUTPUT_DIR, filename=\"best_unet_model\", save_top_k=1, mode=\"min\")\nearly_cb = EarlyStopping(monitor=\"val_loss\", patience=1, mode=\"min\")\n\ntrainer = L.Trainer(\n    max_epochs=15,\n    accelerator=\"auto\",\n    devices=1,\n    precision=\"16-mixed\",\n    callbacks=[ckpt_cb, early_cb],\n    logger=csv_logger,\n    enable_progress_bar=False,\n    log_every_n_steps=10        \n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-03T12:00:15.052162Z","iopub.execute_input":"2026-07-03T12:00:15.052620Z","iopub.status.idle":"2026-07-03T12:00:15.124837Z","shell.execute_reply.started":"2026-07-03T12:00:15.052585Z","shell.execute_reply":"2026-07-03T12:00:15.124255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load model separately — trainer just needs to exist for the dashboard\nloaded_model = MicroscopeSegmentationEngine.load_from_checkpoint(\n    os.path.join(OUTPUT_DIR, \"best_unet_model-v3.ckpt\")\n)\n\nprint(\"✅ Model loaded from checkpoint\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-03T12:00:42.751254Z","iopub.execute_input":"2026-07-03T12:00:42.751531Z","iopub.status.idle":"2026-07-03T12:00:44.299687Z","shell.execute_reply.started":"2026-07-03T12:00:42.751507Z","shell.execute_reply":"2026-07-03T12:00:44.298926Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Complete Dashboard (Curves, Heatmap & Sample Test)","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"id":"fa7a8c9b"}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport torch\nfrom sklearn.metrics import classification_report, confusion_matrix\nfrom torch.utils.data import DataLoader\n\ndef plot_training_and_confusion_dashboard(lightning_model, base_val_dataset, trainer_instance=None):\n    \"\"\"Compiles learning curves and computes a pixel-level normalized confusion matrix \n    using a synchronous single-threaded DataLoader to completely eliminate worker crashes.\n    \"\"\"\n    # 1. Recreate a safe, single-threaded loader instantly to bypass multiprocess deadlocks\n    safe_loader = DataLoader(\n        base_val_dataset, \n        batch_size=32, \n        shuffle=False, \n        num_workers=0, \n        pin_memory=False\n    )\n    \n    metrics_df = None\n    print(\"🔍 [Phase 1] Searching for training convergence history...\")\n\n    # --- Step A: Check standard PyTorch Lightning Trainer Logger History ---\n    if trainer_instance is not None:\n        try:\n            if hasattr(trainer_instance, 'logger') and trainer_instance.logger is not None:\n                # Target the underlying history metrics log array\n                if hasattr(trainer_instance.logger, 'history'):\n                    metrics_df = pd.DataFrame(trainer_instance.logger.history)\n                elif hasattr(trainer_instance.logger, 'experiment') and hasattr(trainer_instance.logger.experiment, 'metrics'):\n                    metrics_df = pd.DataFrame(trainer_instance.logger.experiment.metrics)\n                elif hasattr(trainer_instance, 'logged_metrics') and trainer_instance.logged_metrics:\n                    metrics_df = pd.DataFrame([trainer_instance.logged_metrics])\n                \n                if metrics_df is not None and not metrics_df.empty:\n                    print(\"📊 Metrics history found in active trainer memory!\")\n        except Exception as e:\n            print(f\"⚠️ Memory extraction check skipped: {e}\")\n\n    # --- Step B: Disk Sweep Fallback ---\n    if metrics_df is None or metrics_df.empty:\n        metrics_path = os.path.join(OUTPUT_DIR, \"lightning_logs\")\n        if os.path.exists(metrics_path):\n            versions = sorted(\n                [d for d in os.listdir(metrics_path) if d.startswith(\"version_\")],\n                key=lambda x: int(x.split(\"_\")[1]) if \"_\" in x else 0,\n                reverse=True\n            )\n            for version in versions:\n                metrics_csv = os.path.join(metrics_path, version, \"metrics.csv\")\n                if os.path.exists(metrics_csv):\n                    test_df = pd.read_csv(metrics_csv)\n                    if not test_df.empty:\n                        metrics_df = test_df\n                        print(f\"🎯 Metrics history found on disk: {metrics_csv}\")\n                        break\n\n    # --- Step C: Render Loss & Accuracy as Separate Figures ---\n    if metrics_df is not None and not metrics_df.empty:\n        if 'epoch' in metrics_df.columns:\n            grouped_metrics = metrics_df.groupby('epoch').mean().reset_index()\n\n            # --- Figure: Loss & Accuracy Side by Side ---\n            fig, axes = plt.subplots(1, 2, figsize=(16, 5))\n\n            # Loss\n            if 'train_loss_epoch' in grouped_metrics.columns:\n                axes[0].plot(grouped_metrics['epoch'], grouped_metrics['train_loss_epoch'], label='train_loss', marker='o', color='blue')\n            if 'val_loss' in grouped_metrics.columns:\n                axes[0].plot(grouped_metrics['epoch'], grouped_metrics['val_loss'], label='test_loss', marker='s', color='orange')\n            axes[0].set_title(\"Loss\")\n            axes[0].set_xlabel(\"Epochs\")\n            axes[0].set_ylabel(\"Loss Score\")\n            axes[0].legend()\n            axes[0].grid(True, linestyle='--', alpha=0.6)\n\n            # Accuracy\n            if 'train_acc_epoch' in grouped_metrics.columns:\n                t_acc = grouped_metrics['train_acc_epoch'].dropna()\n                axes[1].plot(grouped_metrics['epoch'], t_acc * 100, color='blue', marker='o', label='train_accuracy')\n            if 'val_acc' in grouped_metrics.columns:\n                v_acc = grouped_metrics['val_acc'].dropna()\n                axes[1].plot(grouped_metrics['epoch'], v_acc * 100, color='orange', marker='s', label='test_accuracy')\n            axes[1].set_title(\"Accuracy\")\n            axes[1].set_xlabel(\"Epochs\")\n            axes[1].set_ylabel(\"Accuracy (%)\")\n            axes[1].legend()\n            axes[1].grid(True, linestyle='--', alpha=0.6)\n\n            plt.tight_layout()\n            plt.show()\n\n        else:\n            print(\"⚠️ Metrics loaded, but missing epoch column. Rendering skipped.\")\n    else:\n        print(\"❌ Could not map learning curves: metrics data missing.\")\n\n    # --- Part 2: Pixel-Level Normalized Confusion Matrix Heatmap Track ---\n    print(\"\\n📊 [Phase 2] Evaluating validation matrix for pixel-level class correlations (Safely using main thread)...\")\n    lightning_model.eval()\n    device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n    lightning_model.to(device)\n\n    all_preds, all_targets = [], []\n\n    with torch.no_grad():\n        for images, masks in safe_loader:\n            logits = lightning_model(images.to(device))\n            preds = torch.argmax(logits, dim=1).cpu().numpy()\n\n            all_preds.append(preds.flatten())\n            all_targets.append(masks.numpy().flatten())\n\n    all_preds = np.concatenate(all_preds)\n    all_targets = np.concatenate(all_targets)\n\n    target_names = ['Background (0)', 'Cells (1)', 'Nuclei (2)']\n    print(\"\\n📝 Pixel-Level Comprehensive Classification Report:\")\n    print(classification_report(all_targets, all_preds, target_names=target_names, labels=[0, 1, 2]))\n\n    cm = confusion_matrix(all_targets, all_preds, normalize='true', labels=[0, 1, 2])\n\n    plt.figure(figsize=(8, 6))\n    sns.heatmap(cm, annot=True, fmt=\".2f\", xticklabels=target_names, yticklabels=target_names, cmap=\"Blues\", cbar=True)\n    plt.title(\"Normalized Confusion Matrix Heatmap (Pixel Relations Layer)\")\n    plt.xlabel(\"Predicted Class Mapping\")\n    plt.ylabel(\"True Ground Truth Class\")\n    plt.tight_layout()\n    plt.show()\n\ndef run_live_sample_test(lightning_model, base_val_dataset, num_samples=2):\n    \"\"\"Executes verification display slices safely using a standard dataset lookup.\"\"\"\n    lightning_model.eval()\n    device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n    lightning_model.to(device)\n    \n    safe_loader = DataLoader(base_val_dataset, batch_size=num_samples, shuffle=False, num_workers=0)\n    images, masks = next(iter(safe_loader))\n    \n    with torch.no_grad():\n        logits = lightning_model(images.to(device))\n        preds = torch.argmax(logits, dim=1).cpu().numpy()\n\n    print(f\"\\n🧪 Running quick side-by-side inference test on {num_samples} slices...\")\n    for i in range(min(num_samples, len(images))):\n        fig, axes = plt.subplots(1, 3, figsize=(18, 6))\n\n        img_np = images[i].permute(1, 2, 0).numpy()\n        rgb_view = np.clip(img_np[:, :, [1, 3, 0]], 0, 1)  # Mapping Microtubules, Target, Nuclei\n\n        axes[0].imshow(rgb_view)\n        axes[0].set_title(f\"Sample {i+1}: Multi-spectral Channels RGB Composite\")\n        axes[0].axis('off')\n\n        axes[1].imshow(masks[i].numpy(), cmap='tab10', vmin=0, vmax=2)\n        axes[1].set_title(\"Ground Truth Class Mask\\n(0: Back, 1: Cells, 2: Nuclei)\")\n        axes[1].axis('off')\n\n        axes[2].imshow(preds[i], cmap='tab10', vmin=0, vmax=2)\n        axes[2].set_title(\"UNet Predicted Morphological Output Mask\")\n        axes[2].axis('off')\n\n        plt.tight_layout()\n        plt.show()","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"id":"ecac639d","trusted":true,"execution":{"iopub.status.busy":"2026-07-03T12:00:47.984586Z","iopub.execute_input":"2026-07-03T12:00:47.985051Z","iopub.status.idle":"2026-07-03T12:00:48.005574Z","shell.execute_reply.started":"2026-07-03T12:00:47.985019Z","shell.execute_reply":"2026-07-03T12:00:48.004768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Run the safe, error-free dashboard panel\nplot_training_and_confusion_dashboard(loaded_model, valid_ds, trainer_instance=trainer)\n\n# 2. Run the slice view checker\nrun_live_sample_test(loaded_model, valid_ds, num_samples=2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-03T12:00:52.047227Z","iopub.execute_input":"2026-07-03T12:00:52.047651Z","iopub.status.idle":"2026-07-03T12:05:26.287674Z","shell.execute_reply.started":"2026-07-03T12:00:52.047619Z","shell.execute_reply":"2026-07-03T12:05:26.286459Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Morphological Post-Processing & Phenotypic Discovery","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"id":"7347c145"}},{"cell_type":"code","source":"class ProductionPhenotypeProfiler:\n    def __init__(self):\n        self.kernel = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (3, 3))\n\n    def clean_mask(self, mask):\n        \"\"\"Filters high-frequency pixel artifacts and closes internal structures.\"\"\"\n        binary_mask = (mask > 0).astype(np.uint8)\n        cleaned = cv2.morphologyEx(binary_mask, cv2.MORPH_OPEN, self.kernel)\n        return cv2.morphologyEx(cleaned, cv2.MORPH_CLOSE, self.kernel)\n\n    def extract_features(self, raw_mask, green_channel):\n        refined = self.clean_mask(raw_mask)\n        props = regionprops(label(refined), intensity_image=green_channel)\n        return pd.DataFrame([{\"cell_id\": p.label, \"area\": p.area, \"mean_intensity\": p.mean_intensity} for p in props])\n\nprofiler = ProductionPhenotypeProfiler()\nprint(\"🎯 Morphological Filtering Engine Initialized.\")","metadata":{"execution":{"iopub.status.busy":"2026-07-03T12:08:07.872521Z","iopub.execute_input":"2026-07-03T12:08:07.873179Z","iopub.status.idle":"2026-07-03T12:08:07.881178Z","shell.execute_reply.started":"2026-07-03T12:08:07.873144Z","shell.execute_reply":"2026-07-03T12:08:07.880456Z"},"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"id":"9d04e24b","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Production Model Compilation (ONNX Conversion)","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"id":"57f9c573"}},{"cell_type":"code","source":"!pip install -q onnxscript","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-03T12:08:22.169707Z","iopub.execute_input":"2026-07-03T12:08:22.170143Z","iopub.status.idle":"2026-07-03T12:08:27.127273Z","shell.execute_reply.started":"2026-07-03T12:08:22.170111Z","shell.execute_reply":"2026-07-03T12:08:27.126158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def export_pipeline_to_onnx():\n    engine = MicroscopeSegmentationEngine.load_from_checkpoint(\n        os.path.join(OUTPUT_DIR, \"best_unet_model-v3.ckpt\")\n    )\n    engine.eval()\n    engine.cpu()\n    onnx_path = os.path.join(OUTPUT_DIR, \"microscope_unet_512.onnx\")\n\n    torch.onnx.export(\n        engine.net,\n        torch.randn(1, 4, 512, 512),\n        onnx_path,\n        export_params=True,\n        opset_version=11,\n        do_constant_folding=True,\n        input_names=['input_channels'],\n        output_names=['segmentation_output'],\n        dynamic_axes={'input_channels': {0: 'batch_size'}, 'segmentation_output': {0: 'batch_size'}}\n    )\n    print(f\"✅ ONNX Production Model Saved to: {onnx_path}\")\n\nexport_pipeline_to_onnx()","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"id":"35e6283f","trusted":true,"execution":{"iopub.status.busy":"2026-07-03T12:08:30.193772Z","iopub.execute_input":"2026-07-03T12:08:30.194621Z","iopub.status.idle":"2026-07-03T12:08:38.456852Z","shell.execute_reply.started":"2026-07-03T12:08:30.194571Z","shell.execute_reply":"2026-07-03T12:08:38.456184Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## FastAPI Microservice Writer (Complete Backend Script)","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"id":"ee269fee"}},{"cell_type":"code","source":"fastapi_script = \"\"\"\nfrom fastapi import FastAPI, UploadFile, File\nimport onnxruntime as ort\nimport numpy as np\nimport cv2\n\napp = FastAPI(title=\"🔬 Microscope AI Production API (Drug Discovery Team)\")\ntry:\n    ort_session = ort.InferenceSession(\"microscope_unet_512.onnx\", providers=['CPUExecutionProvider'])\nexcept Exception:\n    ort_session = None\n\ndef apply_clahe_per_channel(img_fused):\n    clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))\n    equalized = np.zeros_like(img_fused)\n    for c in range(img_fused.shape[2]):\n        ch = img_fused[:, :, c]\n        if ch.dtype != np.uint8:\n            ch = cv2.normalize(ch, None, 0, 255, cv2.NORM_MINMAX).astype(np.uint8)\n        equalized[:, :, c] = clahe.apply(ch)\n    return equalized\n\ndef min_max_normalize(img_fused):\n    img_fused = img_fused.astype(np.float32)\n    for c in range(img_fused.shape[2]):\n        min_val = img_fused[:, :, c].min()\n        max_val = img_fused[:, :, c].max()\n        if (max_val - min_val) > 0:\n            img_fused[:, :, c] = (img_fused[:, :, c] - min_val) / (max_val - min_val)\n        else:\n            img_fused[:, :, c] = 0.0\n    return img_fused\n\ndef morph_clean_mask(mask):\n    kernel = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (3, 3))\n    binary_mask = (mask > 0).astype(np.uint8)\n    cleaned = cv2.morphologyEx(binary_mask, cv2.MORPH_OPEN, kernel)\n    return cv2.morphologyEx(cleaned, cv2.MORPH_CLOSE, kernel)\n\n@app.post(\"/predict\")\nasync def predict_microscope(\n    channel1: UploadFile = File(...), channel2: UploadFile = File(...),\n    channel3: UploadFile = File(...), channel4: UploadFile = File(...)\n):\n    if ort_session is None: return {\"error\": \"Model runtime is offline.\"}\n    channels = []\n    for f in [channel1, channel2, channel3, channel4]:\n        b = await f.read()\n        img = cv2.imdecode(np.frombuffer(b, np.uint8), cv2.IMREAD_GRAYSCALE)\n        channels.append(cv2.resize(img, (256, 256)))\n\n    fused = np.stack(channels, axis=-1)\n    fused = min_max_normalize(apply_clahe_per_channel(fused))\n    tensor = np.expand_dims(np.transpose(fused, (2, 0, 1)), axis=0).astype(np.float32)\n\n    out = ort_session.run(None, {ort_session.get_inputs()[0].name: tensor})\n    raw_mask = np.argmax(out[0], axis=1)[0]\n    refined_mask = morph_clean_mask(raw_mask)\n\n    return {\n        \"status\": \"success\",\n        \"prediction_shape\": list(refined_mask.shape),\n        \"mask_sample\": refined_mask[:5, :5].tolist()\n    }\n\"\"\"\nwith open(\"app.py\", \"w\") as f:\n    f.write(fastapi_script)\nprint(\"✅ Production app.py backend compiled smoothly.\")","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"id":"d5f2c6a2","trusted":true,"execution":{"iopub.status.busy":"2026-07-03T12:08:58.417857Z","iopub.execute_input":"2026-07-03T12:08:58.418575Z","iopub.status.idle":"2026-07-03T12:08:58.426250Z","shell.execute_reply.started":"2026-07-03T12:08:58.418534Z","shell.execute_reply":"2026-07-03T12:08:58.425474Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Streamlit Interface Writer (User Interface Dashboard)","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"id":"84b52bd4"}},{"cell_type":"code","source":"streamlit_code = '''\nimport streamlit as st\nimport numpy as np\nimport cv2\nimport onnxruntime as ort\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as mpatches\nimport io\nimport os\nfrom PIL import Image\nfrom google import genai\nfrom google.genai import types\nfrom datetime import datetime\n\n# ==============================================================================\n# 1. PAGE CONFIGURATION & SECURE API INITIALIZATION\n# ==============================================================================\nst.set_page_config(\n    page_title=\"Microscope AI — Drug Discovery Profiler\",\n    layout=\"wide\",\n    initial_sidebar_state=\"collapsed\"\n)\n\n# Secure Environment Keys Management fetched directly from Space Secrets\nGEMINI_API_KEY = os.environ.get(\"GEMINI_API_KEY\", \"\")\n\n# Initialize modern Google GenAI Client\nclient = None\nif GEMINI_API_KEY:\n    client = genai.Client()\n\n# Load custom CSS Styling optimized for strict vertical stack containment\nst.markdown(\"\"\"\n<style>\n    .stApp { background-color: #0d1117; color: #e6edf3; }\n    .hero-title { font-family: 'Courier New', monospace; font-size: 2.2rem; font-weight: 700; color: #58a6ff; letter-spacing: 0.04em; margin-bottom: 0.2rem; }\n    .hero-sub { font-size: 0.95rem; color: #8b949e; margin-bottom: 2rem; font-family: monospace; }\n    .section-label { font-family: monospace; font-size: 0.85rem; letter-spacing: 0.12em; color: #58a6ff; text-transform: uppercase; margin-top: 2rem; margin-bottom: 0.6rem; border-bottom: 1px solid #21262d; padding-bottom: 0.3rem; }\n    .phase-header { font-family: monospace; font-size: 1.1rem; font-weight: 700; color: #e6edf3; margin-bottom: 0.5rem; margin-top: 1rem; }\n    .stat-card { background: #161b22; border: 1px solid #30363d; border-radius: 8px; padding: 1rem 1.2rem; text-align: center; margin-bottom: 0.5rem; }\n    .status-running { display: inline-block; width: 8px; height: 8px; background: #3fb950; border-radius: 50%; margin-right: 6px; animation: pulse 2s infinite; }\n    @keyframes pulse { 0%, 100% { opacity: 1; } 50% { opacity: 0.4; } }\n    .result-header { font-family: monospace; font-size: 0.85rem; color: #8b949e; text-transform: uppercase; letter-spacing: 0.1em; padding-bottom: 0.4rem; margin-bottom: 0.6rem; }\n    \n    /* Enforce strict uniform monospace alignment on file upload labels */\n    div[data-testid=\"stWidgetLabel\"] p {\n        font-family: monospace !important;\n        font-size: 0.8rem !important;\n        color: #8b949e !important;\n    }\n    div[data-testid=\"stFileUploader\"] { background: #161b22; border: 1px dashed #30363d; border-radius: 8px; padding: 0.5rem; }\n    hr { border-color: #21262d; }\n</style>\n\"\"\", unsafe_allow_html=True)\n\nst.markdown('<div class=\"hero-title\">🔬 Microscope AI — Drug Effect Analysis Profiler</div>', unsafe_allow_html=True)\nst.markdown('<div class=\"hero-sub\">Human Protein Atlas · UNet ResNet34 · 4-Channel Fluorescence Microscopy · Drug Discovery Pipeline</div>', unsafe_allow_html=True)\nst.markdown(\"---\")\n\n@st.cache_resource\ndef load_session():\n    try:\n        return ort.InferenceSession(\"microscope_unet_512.onnx\", providers=['CPUExecutionProvider'])\n    except Exception:\n        return None\n\nsession = load_session()\n\nif session is not None:\n    st.markdown('<span class=\"status-running\"></span><span style=\"font-family:monospace;font-size:0.8rem;color:#3fb950;\">ONNX Inference Engine — Online</span>', unsafe_allow_html=True)\nelse:\n    st.error(\"⚠️ ONNX model failed to load.\")\n\nst.markdown(\"<br>\", unsafe_allow_html=True)\n\n# ==============================================================================\n# 2. THE HPA KNOWLEDGE BASE & SYSTEM INSTRUCTIONS (THE CORPUS)\n# ==============================================================================\nHPA_KNOWLEDGE_BASE = \"\"\"HUMAN PROTEIN ATLAS (HPA) DIGITAL PATHOLOGY CORPUS - PRODUCTION GRADE\n1. SYSTEMS ARCHITECTURE & IMAGE CHANNELS:\n- Input Protocol: Four synchronized, single-channel grayscale images normalized to [0, 255] representing:\n  * BLUE Channel (DAPI): Stains Nuclear boundary, Nuclear Area, Nucleoli cavities.\n  * RED Channel (Anti-Tubulin): Stains Microtubules. Maps cell structural morphology and volume.\n  * YELLOW Channel (Anti-Calreticulin): Stains Endoplasmic Reticulum (ER). Defines internal cytoplasmic networks.\n  * GREEN Channel (Target Antibody): Stains unknown protein under investigation.\"\"\"\n\nSYSTEM_INSTRUCTION = \"\"\"ROLE & IDENTITY:\nYou are \"Microscope AI Expert Co-Pilot\", a cutting-edge Multimodal AI Medical Pathologist specialized in the Human Protein Atlas (HPA) and drug morphology interactions.\nOPERATIONAL MANDATE:\nEvaluate the multi-spectral variance values reported by the U-Net workspace. Always respond in the language used by the researcher. If they ask in Arabic, use professional Arabic medical terminology while keeping English organelle names intact.\"\"\"\n\ndef generate_gemini_content(prompt, pil_image=None):\n    if not client:\n        return \"⚠️ GEMINI_API_KEY is missing in Space Secrets/Environment Variables!\"\n    try:\n        contents = []\n        if pil_image:\n            contents.append(pil_image)\n        contents.append(prompt)\n        response = client.models.generate_content(\n            model=\"gemini-2.5-flash\",\n            contents=contents,\n            config=types.GenerateContentConfig(\n                system_instruction=SYSTEM_INSTRUCTION,\n                temperature=0.3,\n            )\n        )\n        return response.text\n    except Exception as e:\n        return \"❌ Failed to execute Gemini API Request: \" + str(e)\n\n# ==============================================================================\n# 3. CORE SEGMENTATION PIPELINE FUNCTIONS\n# ==============================================================================\ndef apply_clahe_per_channel(img_fused):\n    clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))\n    equalized = np.zeros_like(img_fused)\n    for c in range(img_fused.shape[2]):\n        ch = img_fused[:, :, c]\n        if ch.dtype != np.uint8:\n            ch = cv2.normalize(ch, None, 0, 255, cv2.NORM_MINMAX).astype(np.uint8)\n        equalized[:, :, c] = clahe.apply(ch)\n    return equalized\n\ndef min_max_normalize(img_fused):\n    img_fused = img_fused.astype(np.float32)\n    for c in range(img_fused.shape[2]):\n        min_val = img_fused[:, :, c].min()\n        max_val = img_fused[:, :, c].max()\n        if (max_val - min_val) > 0:\n            img_fused[:, :, c] = (img_fused[:, :, c] - min_val) / (max_val - min_val)\n        else:\n            img_fused[:, :, c] = 0.0\n    return img_fused\n\ndef run_inference(uploaded_dict, keys):\n    channels = []\n    for key in keys:\n        raw = np.frombuffer(uploaded_dict[key].read(), np.uint8)\n        img = cv2.imdecode(raw, cv2.IMREAD_GRAYSCALE)\n        channels.append(cv2.resize(img, (512, 512)))\n    fused = np.stack(channels, axis=-1)\n    fused = min_max_normalize(apply_clahe_per_channel(fused))\n    tensor = np.expand_dims(np.transpose(fused, (2, 0, 1)), axis=0).astype(np.float32)\n    out = session.run(None, {session.get_inputs()[0].name: tensor})\n    mask = np.argmax(out[0][0], axis=0)\n    return fused, mask\n\ndef get_stats(mask):\n    total = mask.size\n    bg_px   = int(np.sum(mask == 0))\n    cell_px = int(np.sum(mask == 1))\n    nuc_px  = int(np.sum(mask == 2))\n    return {\n        \"bg_pct\":   bg_px / total * 100,\n        \"cell_pct\": cell_px / total * 100,\n        \"nuc_pct\":  nuc_px / total * 100,\n        \"ratio\":    (cell_px / nuc_px) if nuc_px > 0 else 0,\n        \"cell_px\":  cell_px,\n        \"nuc_px\":   nuc_px,\n        \"bg_px\":    bg_px,\n    }\n\ndef render_mask(mask, stats):\n    COLOR_MAP = {0: [13,17,23], 1: [56,189,248], 2: [249,115,22]}\n    mask_rgb = np.zeros((*mask.shape, 3), dtype=np.uint8)\n    for cls, color in COLOR_MAP.items():\n        mask_rgb[mask == cls] = color\n    fig, ax = plt.subplots(figsize=(5, 5))\n    fig.patch.set_facecolor('#0d1117')\n    ax.set_facecolor('#0d1117')\n    ax.imshow(mask_rgb)\n    ax.axis('off')\n    patches = [\n        mpatches.Patch(color=np.array(COLOR_MAP[0])/255, label='Background (' + str(round(stats[\"bg_pct\"], 1)) + '%)'),\n        mpatches.Patch(color=np.array(COLOR_MAP[1])/255, label='Cell Body (' + str(round(stats[\"cell_pct\"], 1)) + '%)'),\n        mpatches.Patch(color=np.array(COLOR_MAP[2])/255, label='Nucleus (' + str(round(stats[\"nuc_pct\"], 1)) + '%)'),\n    ]\n    ax.legend(handles=patches, loc='lower left', fontsize=8, framealpha=0.85,\n              facecolor='#161b22', edgecolor='#30363d', labelcolor='#e6edf3')\n    return fig\n\n# ==============================================================================\n# 4. FRONTEND STREAMLIT VERTICAL ARTIFACT PIPELINE STACK\n# ==============================================================================\nCHANNEL_LABELS = [\"🔵 Blue — Nuclei\", \"🔴 Red — Microtubules\", \"🟡 Yellow — ER\", \"🟢 Green — Target Protein\"]\nCHANNEL_KEYS   = [\"blue\", \"red\", \"yellow\", \"green\"]\n\n# --- BLOCK A: MULTI-CHANNEL DATA INPUT HUB ---\nst.markdown('<div class=\"section-label\">1. Assay Sample Upload Matrix</div>', unsafe_allow_html=True)\n\nst.markdown('<div class=\"phase-header\">📋 Control Group — Before Drug Treatment</div>', unsafe_allow_html=True)\nbefore = {}\nb_cols = st.columns(4)\nfor col, label, key in zip(b_cols, CHANNEL_LABELS, CHANNEL_KEYS):\n    with col:\n        before[key] = st.file_uploader(label, type=[\"png\",\"jpg\"], key=\"before_\" + key, label_visibility=\"visible\")\n\nst.markdown('<div class=\"phase-header\">💊 Experimental Group — After Drug Treatment</div>', unsafe_allow_html=True)\nafter = {}\na_cols = st.columns(4)\nfor col, label, key in zip(a_cols, CHANNEL_LABELS, CHANNEL_KEYS):\n    with col:\n        after[key] = st.file_uploader(label, type=[\"png\",\"jpg\"], key=\"after_\" + key, label_visibility=\"visible\")\n\nbefore_ready = all(before[k] is not None for k in CHANNEL_KEYS)\nafter_ready  = all(after[k] is not None for k in CHANNEL_KEYS)\n\n# --- BLOCK B: SEGMENTATION MASK GENERATION INTERFACE ---\nif before_ready and after_ready and session is not None:\n    with st.spinner(\"Processing deep learning segmentations down the stack...\"):\n        fused_b, mask_b = run_inference(before, CHANNEL_KEYS)\n        fused_a, mask_a = run_inference(after,  CHANNEL_KEYS)\n        stats_b = get_stats(mask_b)\n        stats_a = get_stats(mask_a)\n        st.session_state[\"pipeline_stats\"] = {\"before\": stats_b, \"after\": stats_a}\n\n    st.markdown('<div class=\"section-label\">2. UNet Segmentation & Morphological Maps</div>', unsafe_allow_html=True)\n    \n    # Render Control Row\n    st.markdown('<div class=\"phase-header\">📋 Baseline Control Visualizer</div>', unsafe_allow_html=True)\n    vis_b_left, vis_b_right = st.columns(2)\n    with vis_b_left:\n        st.markdown('<div class=\"result-header\">Control Sample — Composite Overlay</div>', unsafe_allow_html=True)\n        st.image(np.clip(fused_b[:, :, [1, 3, 0]], 0, 1), use_container_width=True)\n    with vis_b_right:\n        st.markdown('<div class=\"result-header\">Control Sample — Predicted Mask</div>', unsafe_allow_html=True)\n        st.pyplot(render_mask(mask_b, stats_b), use_container_width=True)\n\n    # Render Treated Row\n    st.markdown('<div class=\"phase-header\">💊 Drug-Treated Experimental Visualizer</div>', unsafe_allow_html=True)\n    vis_a_left, vis_a_right = st.columns(2)\n    with vis_a_left:\n        st.markdown('<div class=\"result-header\">Treated Sample — Composite Overlay</div>', unsafe_allow_html=True)\n        st.image(np.clip(fused_a[:, :, [1, 3, 0]], 0, 1), use_container_width=True)\n    with vis_a_right:\n        st.markdown('<div class=\"result-header\">Treated Sample — Predicted Mask</div>', unsafe_allow_html=True)\n        st.pyplot(render_mask(mask_a, stats_a), use_container_width=True)\n\n    # --- BLOCK C: QUANTITATIVE ANALYTICS HUB ---\n    st.markdown('<div class=\"section-label\">3. Quantitative Morphological Assay Comparison</div>', unsafe_allow_html=True)\n    metrics = [\n        (\"Cell Body Coverage\", stats_b[\"cell_pct\"], stats_a[\"cell_pct\"], \"%\"),\n        (\"Nucleus Coverage\",   stats_b[\"nuc_pct\"],  stats_a[\"nuc_pct\"],  \"%\"),\n        (\"Background Density\",  stats_b[\"bg_pct\"],   stats_a[\"bg_pct\"],   \"%\"),\n        (\"Cell / Nucleus Ratio\", stats_b[\"ratio\"],  stats_a[\"ratio\"],    \"x\"),\n    ]\n    c1, c2, c3, c4 = st.columns(4)\n    for col, (label, before_val, after_val, unit) in zip([c1,c2,c3,c4], metrics):\n        delta = after_val - before_val\n        with col:\n            st.markdown('<div class=\"stat-card\"><div style=\"font-size:0.7rem;color:#8b949e;font-family:monospace;text-transform:uppercase;margin-bottom:0.5rem;\">' + label + '</div><div style=\"font-family:monospace;font-size:0.9rem;color:#8b949e;\">Before: <b style=\"color:#e6edf3\">' + str(round(before_val, 1)) + unit + '</b></div><div style=\"font-family:monospace;font-size:0.9rem;color:#8b949e;\">After: <b style=\"color:#58a6ff\">' + str(round(after_val, 1)) + unit + '</b></div></div>', unsafe_allow_html=True)\n\n# --- BLOCK D: FULL WIDTH PATHOLOGIST CO-PILOT CHATBOT HUB ---\nst.markdown('<div class=\"section-label\">4. Microscope AI Pathologist Co-Pilot Chat System</div>', unsafe_allow_html=True)\n\nif \"messages\" not in st.session_state:\n    st.session_state.messages = [{\"role\": \"assistant\", \"content\": \"Natively synchronized with backend profiling node. Upload files to instantiate generative feedback loops.\"}]\n\nif \"pipeline_stats\" in st.session_state and len(st.session_state.messages) == 1:\n    st_b = st.session_state[\"pipeline_stats\"][\"before\"]\n    st_a = st.session_state[\"pipeline_stats\"][\"after\"]\n    \n    report_prompt = \"Write an academic grade microscopic assay pathology report based on these observations: Control sample cell body coverage is \" + str(round(st_b['cell_pct'], 2)) + \"% and nucleus coverage is \" + str(round(st_b['nuc_pct'], 2)) + \"%. Post drug-doping treatment, cell body coverage shifted to \" + str(round(st_a['cell_pct'], 2)) + \"% and nucleus coverage shifted to \" + str(round(st_a['nuc_pct'], 2)) + \"%.\"\n    \n    with st.spinner(\"Generating automated pathology assessment matrix down the baseline stack...\"):\n        response_text = generate_gemini_content(report_prompt)\n        st.session_state.messages.append({\"role\": \"assistant\", \"content\": response_text})\n\n# Render conversation logs across full content container width\nfor msg in st.session_state.messages:\n    with st.chat_message(msg[\"role\"]):\n        st.write(msg[\"content\"])\n\nif user_query := st.chat_input(\"Query cell morphology metrics across the stack...\"):\n    with st.chat_message(\"user\"):\n        st.write(user_query)\n    st.session_state.messages.append({\"role\": \"user\", \"content\": user_query})\n\n    ctx_stats = str(st.session_state.get(\"pipeline_stats\", \"No active segmentations run yet.\"))\n    full_p = \"Context Metrics: \" + ctx_stats + \" | Researcher Query: \" + str(user_query)\n\n    with st.chat_message(\"assistant\"):\n        with st.spinner(\"Reviewing knowledge base archives...\"):\n            res_text = generate_gemini_content(full_p)\n            st.write(res_text)\n    st.session_state.messages.append({\"role\": \"assistant\", \"content\": res_text})\n'''\n\nwith open(\"/kaggle/working/streamlit_app.py\", \"w\") as f:\n    f.write(streamlit_code)\nprint(\"✅ Streamlit UI Code file successfully saved down vertically stacked.\")","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"id":"433a375e","trusted":true,"execution":{"iopub.status.busy":"2026-07-03T14:19:12.645671Z","iopub.execute_input":"2026-07-03T14:19:12.646022Z","iopub.status.idle":"2026-07-03T14:19:12.659770Z","shell.execute_reply.started":"2026-07-03T14:19:12.645989Z","shell.execute_reply":"2026-07-03T14:19:12.659039Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### vers","metadata":{}},{"cell_type":"code","source":"import pkg_resources\n\npackages = [\n    'streamlit', 'torch', 'torchvision', 'segmentation-models-pytorch',\n    'opencv-python-headless', 'numpy', 'pandas', 'scikit-image',\n    'onnxruntime', 'lightning', 'torchmetrics', 'albumentations',\n    'google-genai', 'pillow', 'matplotlib'\n]\n\nreqs = []\nfor p in packages:\n    try:\n        version = pkg_resources.get_distribution(p).version\n        if \"+\" in version:\n            version = version.split(\"+\")[0]\n            \n        if p in ['matplotlib', 'numpy', 'pandas']:\n            reqs.append(p)\n            print(f\"⚠️ {p} added unpinned to ensure Python 3.12 image compatibility\")\n        else:\n            reqs.append(f\"{p}=={version}\")\n            print(f\"✅ {p}=={version}\")\n    except Exception:\n        reqs.append(p)\n        print(f\"⚠️ {p} added unpinned\")\n\nwith open(\"/kaggle/working/requirements.txt\", \"w\") as f:\n    f.write(\"\\n\".join(reqs))\n\nprint(\"\\n🎯 Clean requirements.txt successfully saved to your workspace.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-03T13:16:21.642374Z","iopub.execute_input":"2026-07-03T13:16:21.642680Z","iopub.status.idle":"2026-07-03T13:16:21.651191Z","shell.execute_reply.started":"2026-07-03T13:16:21.642653Z","shell.execute_reply":"2026-07-03T13:16:21.650434Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Automated Hugging Face Space Deployment","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"id":"0f7aec48"}},{"cell_type":"code","source":"from huggingface_hub import HfApi\nfrom kaggle_secrets import UserSecretsClient\nimport os\n\nHF_TOKEN = UserSecretsClient().get_secret(\"HF_TOKEN\")\nUSER_ID = \"Jimmy1O\"\nSPACE_NAME = \"microscope-ai-cell-and-chatbot\"\napi = HfApi()\n\n# ── Write Clean Production Dockerfile ──────────────────\nwith open(\"/kaggle/working/Dockerfile\", \"w\") as f:\n    f.write(\"\"\"FROM python:3.12-slim\nWORKDIR /app\n\nRUN apt-get update && apt-get install -y \\\\\n    build-essential \\\\\n    curl \\\\\n    git \\\\\n    libgl1 \\\\\n    libglib2.0-0 \\\\\n    && rm -rf /var/lib/apt/lists/*\n\nCOPY requirements.txt .\n\nRUN pip install --no-cache-dir --upgrade pip setuptools wheel\nRUN pip uninstall -y google-genai google-generativeai || true\n\nRUN pip install --no-cache-dir -r requirements.txt\nCOPY . .\n\nEXPOSE 7860\n\nHEALTHCHECK CMD curl --fail http://localhost:7860/_stcore/health || exit 1\n\nENTRYPOINT [\"streamlit\", \"run\", \"streamlit_app.py\", \"--server.port=7860\", \"--server.address=0.0.0.0\", \"--server.enableCORS=false\", \"--server.enableXsrfProtection=false\"]\n\"\"\")\n\nprint(\"✅ Dedicated Dockerfile saved successfully.\")\n\n# ── Verify requirements are present ─────────────────────────────────────────\nif os.path.exists(\"/kaggle/working/requirements.txt\"):\n    with open(\"/kaggle/working/requirements.txt\", \"r\") as f:\n        content = f.read()\n    print(\"✅ Verified requirements.txt content file matches.\")\nelse:\n    print(\"❌ Error: Run your requirements generator cell first!\")\n\n# ── Write Streamlit config ─────────────────────────────────────────────────────\nos.makedirs(\"/kaggle/working/.streamlit\", exist_ok=True)\nwith open(\"/kaggle/working/.streamlit/config.toml\", \"w\") as f:\n    f.write(\"\"\"[server]\nmaxUploadSize = 200\nenableXsrfProtection = false\nenableCORS = false\n\"\"\")\nprint(\"✅ .streamlit/config.toml updated\")\n\n# ── Push over to Hugging Face Spaces ─────────────────────────────────────────────\ntry:\n    repo_id = f\"{USER_ID}/{SPACE_NAME}\"\n    api.create_repo(\n        repo_id=repo_id,\n        repo_type=\"space\",\n        space_sdk=\"docker\",\n        token=HF_TOKEN,\n        exist_ok=True\n    )\n\n    files = {\n        \"/kaggle/working/streamlit_app.py\":           \"streamlit_app.py\",\n        \"/kaggle/working/requirements.txt\":           \"requirements.txt\",\n        \"/kaggle/working/Dockerfile\":                 \"Dockerfile\",\n        \"/kaggle/working/.streamlit/config.toml\":     \".streamlit/config.toml\",\n        \"/kaggle/working/microscope_unet_512.onnx\":       \"microscope_unet_512.onnx\",\n        \"/kaggle/working/microscope_unet_512.onnx.data\":  \"microscope_unet_512.onnx.data\"\n    }\n\n    print(f\"\\\\n📦 Uploading production artifacts to {repo_id}...\")\n    for local_path, repo_filename in files.items():\n        if os.path.exists(local_path):\n            print(f\"  Uploading {repo_filename}...\")\n            api.upload_file(\n                path_or_fileobj=local_path,\n                path_in_repo=repo_filename,\n                repo_id=repo_id,\n                repo_type=\"space\",\n                token=HF_TOKEN\n            )\n        else:\n            print(f\"  ⚠️ Skipped (not found): {local_path}\")\n\n    print(f\"\\\\n✅ Deployment complete!\")\n    print(f\"🔗 https://huggingface.co/spaces/{repo_id}\")\n\nexcept Exception as e:\n    print(f\"❌ Deployment failed: {e}\")","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"id":"4bb1ca97","trusted":true,"execution":{"iopub.status.busy":"2026-07-03T14:19:20.798221Z","iopub.execute_input":"2026-07-03T14:19:20.799204Z","iopub.status.idle":"2026-07-03T14:19:24.391052Z","shell.execute_reply.started":"2026-07-03T14:19:20.799127Z","shell.execute_reply":"2026-07-03T14:19:24.390350Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Sample tests for hf","metadata":{}},{"cell_type":"code","source":"import shutil\nimport random\n\nTEST_IMG_DIR = \"/kaggle/input/competitions/human-protein-atlas-image-classification/test\"\n\n# Get unique IDs\ntest_files = os.listdir(TEST_IMG_DIR)\ntest_ids = list(set([f.rsplit('_', 1)[0] for f in test_files]))\nprint(f\"Total test images: {len(test_ids)}\")\n\n# Pick a random one every run\nsample_id = random.choice(test_ids)\nprint(f\"Randomly selected ID: {sample_id}\")\n\nchannels = ['blue', 'red', 'yellow', 'green']\nfor ch in channels:\n    src = f\"{TEST_IMG_DIR}/{sample_id}_{ch}.png\"\n    dst = f\"/kaggle/working/sample_{ch}.png\"  # fixed filenames so they always override\n    shutil.copy(src, dst)\n    print(f\"✅ Copied {ch} → sample_{ch}.png\")\n\nprint(f\"\\n📁 4 channels saved to /kaggle/working/ — ready to upload to the app\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-03T00:46:18.331190Z","iopub.execute_input":"2026-07-03T00:46:18.331526Z","iopub.status.idle":"2026-07-03T00:46:19.240630Z","shell.execute_reply.started":"2026-07-03T00:46:18.331496Z","shell.execute_reply":"2026-07-03T00:46:19.239779Z"}},"outputs":[],"execution_count":null}]}