{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":49997,"databundleVersionId":5295352}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# ","metadata":{}},{"cell_type":"markdown","source":"# Early Detection of 3D Printing Defects: From Vision Model to Operator Decision Support\n\n\nIn additive manufacturing, a subtle defect like under-extrusion can silently ruin a multi-hour print. So can we detect if a print is trending toward failure early enough to intervene?\"\n\nWhat this notebook builds\n1.  Detection: Fine-tuned MobileNetV2 on nozzle-camera frames from 7 printers. Printer-aware split (unseen printers in validation), weighted sampling for class imbalance, evaluated on PR-AUC.\n2.  Decision Support (zero extra training):\n   - Early warning — How early in a defective print does the system catch it?\n   - Health score (0–100) — Real-time operator dashboard metric\n   - Automated alerts — Actionable recommendations\n   - Threshold analysis — The detection vs. false alarm trade-off","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-19T06:32:18.158917Z","iopub.execute_input":"2026-05-19T06:32:18.159136Z","iopub.status.idle":"2026-05-19T06:34:42.791175Z","shell.execute_reply.started":"2026-05-19T06:32:18.159111Z","shell.execute_reply":"2026-05-19T06:34:42.790149Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, random, warnings, gc\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom pathlib import Path\nfrom collections import Counter\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader, WeightedRandomSampler\nfrom torchvision import transforms, models\nfrom PIL import Image\nfrom sklearn.metrics import (\n    classification_report, confusion_matrix, precision_recall_curve,\n    average_precision_score, roc_auc_score, f1_score\n)\nfrom sklearn.model_selection import GroupShuffleSplit\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T06:57:11.092416Z","iopub.execute_input":"2026-05-19T06:57:11.093539Z","iopub.status.idle":"2026-05-19T06:57:20.365229Z","shell.execute_reply.started":"2026-05-19T06:57:11.093502Z","shell.execute_reply":"2026-05-19T06:57:20.364584Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"warnings.filterwarnings(\"ignore\")\nsns.set_theme(style=\"darkgrid\", palette=\"muted\")\nSEED = 42\nrandom.seed(SEED)\nnp.random.seed(SEED)\ntorch.manual_seed(SEED)\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Device: {DEVICE}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T06:57:20.366449Z","iopub.execute_input":"2026-05-19T06:57:20.366964Z","iopub.status.idle":"2026-05-19T06:57:20.622658Z","shell.execute_reply.started":"2026-05-19T06:57:20.366939Z","shell.execute_reply":"2026-05-19T06:57:20.622012Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Load and explore data","metadata":{}},{"cell_type":"code","source":"DATA_ROOT = Path(\"/kaggle/input/competitions/early-detection-of-3d-printing-issues\")\nIMG_DIR   = DATA_ROOT / \"images\"\nTRAIN_CSV = DATA_ROOT / \"train.csv\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T06:57:22.876193Z","iopub.execute_input":"2026-05-19T06:57:22.876781Z","iopub.status.idle":"2026-05-19T06:57:22.880992Z","shell.execute_reply.started":"2026-05-19T06:57:22.876755Z","shell.execute_reply":"2026-05-19T06:57:22.880291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(TRAIN_CSV)\nprint(f\"Total samples : {len(df)}\")\nprint(f\"Columns       : {list(df.columns)}\")\nprint(f\"\\nPrinters      : {df['printer_id'].nunique()}  →  {sorted(df['printer_id'].unique())}\")\nprint(f\"Print jobs    : {df['print_id'].nunique()}\")\nprint(f\"\\nClass balance :\")\nprint(df[\"has_under_extrusion\"].value_counts(normalize=True).to_string())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T06:57:25.272263Z","iopub.execute_input":"2026-05-19T06:57:25.273111Z","iopub.status.idle":"2026-05-19T06:57:25.400605Z","shell.execute_reply.started":"2026-05-19T06:57:25.273079Z","shell.execute_reply":"2026-05-19T06:57:25.399721Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 3, figsize=(18, 5))\n\n\n# explore class imbalance\ncounts = df[\"has_under_extrusion\"].value_counts()\naxes[0].bar([\"Healthy (0)\", \"Under-Extrusion (1)\"], counts.values,\n            color=[\"#2ecc71\", \"#e74c3c\"], edgecolor=\"black\")\naxes[0].set_title(\"Class Distribution\", fontsize=14, fontweight=\"bold\")\nfor i, v in enumerate(counts.values):\n    axes[0].text(i, v + 50, f\"{v}\\n({v/len(df)*100:.1f}%)\", ha=\"center\", fontsize=11)\n\n\n# per printer smaple\nprinter_counts = df.groupby([\"printer_id\", \"has_under_extrusion\"]).size().unstack(fill_value=0)\nprinter_counts.plot(kind=\"bar\", stacked=True, ax=axes[1],\n                    color=[\"#2ecc71\", \"#e74c3c\"], edgecolor=\"black\")\naxes[1].set_title(\"Samples per Printer\", fontsize=14, fontweight=\"bold\")\naxes[1].set_xlabel(\"Printer ID\")\naxes[1].legend([\"Healthy\", \"Under-Extrusion\"])\n\n\n# samples per print job (top 30)\nprint_sizes = df.groupby(\"print_id\").size().sort_values(ascending=False).head(30)\naxes[2].barh(print_sizes.index.astype(str), print_sizes.values, color=\"#3498db\", edgecolor=\"black\")\naxes[2].set_title(\"Frames per Print Job (Top 30)\", fontsize=14, fontweight=\"bold\")\naxes[2].invert_yaxis()\nplt.tight_layout(); plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T06:57:27.436307Z","iopub.execute_input":"2026-05-19T06:57:27.436607Z","iopub.status.idle":"2026-05-19T06:57:28.272733Z","shell.execute_reply.started":"2026-05-19T06:57:27.436585Z","shell.execute_reply":"2026-05-19T06:57:28.272002Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"sample images for viz","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 5, figsize=(20, 8))\nfor row, label in enumerate([0, 1]):\n    subset = df[df[\"has_under_extrusion\"] == label].sample(5, random_state=SEED)\n    for col, (_, r) in enumerate(subset.iterrows()):\n        img = Image.open(IMG_DIR / r[\"img_path\"])\n        axes[row, col].imshow(img)\n        axes[row, col].axis(\"off\")\n    axes[row, 0].set_ylabel(\"Under-Extrusion\" if label else \"Healthy\",\n                            fontsize=14, fontweight=\"bold\", rotation=0, labelpad=100)\nfig.suptitle(\"Sample Frames: Healthy vs. Under-Extrusion\", fontsize=16, fontweight=\"bold\", y=1.02)\nplt.tight_layout(); plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T06:57:29.944524Z","iopub.execute_input":"2026-05-19T06:57:29.945336Z","iopub.status.idle":"2026-05-19T06:57:31.307970Z","shell.execute_reply.started":"2026-05-19T06:57:29.945306Z","shell.execute_reply":"2026-05-19T06:57:31.306942Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Dataset Split","metadata":{}},{"cell_type":"code","source":"gss = GroupShuffleSplit(n_splits=1, test_size=0.25, random_state=SEED)\ntrain_idx, val_idx = next(gss.split(df, groups=df[\"printer_id\"]))\ntrain_df = df.iloc[train_idx].reset_index(drop=True)\nval_df   = df.iloc[val_idx].reset_index(drop=True)\n\nprint(f\"Train : {len(train_df)}  |  Val : {len(val_df)}\")\nprint(f\"Train printers : {sorted(train_df['printer_id'].unique())}\")\nprint(f\"Val   printers : {sorted(val_df['printer_id'].unique())}\")\nprint(f\"\\nTrain class balance:\")\nprint(train_df[\"has_under_extrusion\"].value_counts(normalize=True).to_string())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T06:57:35.976621Z","iopub.execute_input":"2026-05-19T06:57:35.977365Z","iopub.status.idle":"2026-05-19T06:57:35.994675Z","shell.execute_reply.started":"2026-05-19T06:57:35.977334Z","shell.execute_reply":"2026-05-19T06:57:35.993946Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Image preproessing: normalization and resizing","metadata":{}},{"cell_type":"code","source":"IMG_SIZE = 224\ntrain_transforms = transforms.Compose([\n    transforms.Resize((IMG_SIZE, IMG_SIZE)),\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomVerticalFlip(),\n    transforms.RandomRotation(15),\n    transforms.ColorJitter(brightness=0.2, contrast=0.2),\n    transforms.ToTensor(),\n    transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225]),\n])\nval_transforms = transforms.Compose([\n    transforms.Resize((IMG_SIZE, IMG_SIZE)),\n    transforms.ToTensor(),\n    transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225]),\n])\nclass PrintDefectDataset(Dataset):\n    \"\"\"Single-frame dataset for the CNN.\"\"\"\n    def __init__(self, dataframe, img_dir, transform=None):\n        self.df = dataframe\n        self.img_dir = img_dir\n        self.transform = transform\n    def __len__(self):\n        return len(self.df)\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img = Image.open(self.img_dir / row[\"img_path\"]).convert(\"RGB\")\n        if self.transform:\n            img = self.transform(img)\n        label = torch.tensor(row[\"has_under_extrusion\"], dtype=torch.float32)\n        return img, label\n\n\ntrain_ds = PrintDefectDataset(train_df, IMG_DIR, train_transforms)\nval_ds   = PrintDefectDataset(val_df,   IMG_DIR, val_transforms)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T06:57:37.868428Z","iopub.execute_input":"2026-05-19T06:57:37.869358Z","iopub.status.idle":"2026-05-19T06:57:37.877587Z","shell.execute_reply.started":"2026-05-19T06:57:37.869327Z","shell.execute_reply":"2026-05-19T06:57:37.876930Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class_counts = Counter(train_df[\"has_under_extrusion\"].values)\nweights = [1.0 / class_counts[int(l)] for l in train_df[\"has_under_extrusion\"]]\nsampler = WeightedRandomSampler(weights, num_samples=len(weights), replacement=True)\n\nBATCH_SIZE = 32 #first test\n\ntrain_loader = DataLoader(train_ds, batch_size=BATCH_SIZE, sampler=sampler, num_workers=2)\nval_loader   = DataLoader(val_ds,   batch_size=BATCH_SIZE, shuffle=False, num_workers=2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T06:57:39.953484Z","iopub.execute_input":"2026-05-19T06:57:39.953983Z","iopub.status.idle":"2026-05-19T06:57:39.996358Z","shell.execute_reply.started":"2026-05-19T06:57:39.953952Z","shell.execute_reply":"2026-05-19T06:57:39.995675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class StaticCNN(nn.Module):\n    \"\"\"MobileNetV2 backbone with a single-neuron binary head.\"\"\"\n    def __init__(self):\n        super().__init__()\n        backbone = models.mobilenet_v2(weights=\"IMAGENET1K_V1\")\n        # Freeze early layers, fine-tune later ones\n        for i, param in enumerate(backbone.features.parameters()):\n            if i < 100:\n                param.requires_grad = False\n        self.features = backbone.features\n        self.pool = nn.AdaptiveAvgPool2d(1)\n        self.head = nn.Sequential(\n            nn.Dropout(0.3),\n            nn.Linear(1280, 1)\n        )\n    def forward(self, x):\n        x = self.features(x)\n        x = self.pool(x).flatten(1)\n        return self.head(x).squeeze(1)\n        \n    def extract_features(self, x):\n        \"\"\"will use for time series exploration later by the LSTM.\"\"\"\n        x = self.features(x)\n        x = self.pool(x).flatten(1)\n        return x\n        \nmodel_cnn = StaticCNN().to(DEVICE)\nprint(f\"Trainable params: {sum(p.numel() for p in model_cnn.parameters() if p.requires_grad):,}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T06:57:41.313520Z","iopub.execute_input":"2026-05-19T06:57:41.314349Z","iopub.status.idle":"2026-05-19T06:57:41.974278Z","shell.execute_reply.started":"2026-05-19T06:57:41.314307Z","shell.execute_reply":"2026-05-19T06:57:41.973396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"EPOCHS_CNN = 8\nLR = 1e-4\n\n\npos_weight = torch.tensor([class_counts[0] / class_counts[1]]).to(DEVICE)\ncriterion = nn.BCEWithLogitsLoss(pos_weight=pos_weight)\noptimizer = optim.Adam(filter(lambda p: p.requires_grad, model_cnn.parameters()), lr=LR)\nscheduler = optim.lr_scheduler.CosineAnnealingLR(optimizer, T_max=EPOCHS_CNN)\nhistory = {\"train_loss\": [], \"val_loss\": [], \"val_f1\": [], \"val_prauc\": []}\nfor epoch in range(EPOCHS_CNN):\n    # ── Train ──\n    model_cnn.train()\n    running_loss = 0.0\n    for imgs, labels in train_loader:\n        imgs, labels = imgs.to(DEVICE), labels.to(DEVICE)\n        optimizer.zero_grad()\n        logits = model_cnn(imgs)\n        loss = criterion(logits, labels)\n        loss.backward()\n        optimizer.step()\n        running_loss += loss.item() * imgs.size(0)\n    scheduler.step()\n    train_loss = running_loss / len(train_ds)\n    # ── Validate ──\n    model_cnn.eval()\n    val_loss, all_logits, all_labels = 0.0, [], []\n    with torch.no_grad():\n        for imgs, labels in val_loader:\n            imgs, labels = imgs.to(DEVICE), labels.to(DEVICE)\n            logits = model_cnn(imgs)\n            val_loss += criterion(logits, labels).item() * imgs.size(0)\n            all_logits.append(logits.cpu())\n            all_labels.append(labels.cpu())\n    val_loss /= len(val_ds)\n    all_logits = torch.cat(all_logits).numpy()\n    all_labels = torch.cat(all_labels).numpy()\n    probs = 1 / (1 + np.exp(-all_logits))  # sigmoid\n    preds = (probs >= 0.5).astype(int)\n    \n    f1  = f1_score(all_labels, preds)\n    prauc = average_precision_score(all_labels, probs)\n    \n    history[\"train_loss\"].append(train_loss)\n    history[\"val_loss\"].append(val_loss)\n    history[\"val_f1\"].append(f1)\n    history[\"val_prauc\"].append(prauc)\n    print(f\"Epoch {epoch+1}/{EPOCHS_CNN}  |  \"\n          f\"Train Loss: {train_loss:.4f}  |  Val Loss: {val_loss:.4f}  |  \"\n          f\"F1: {f1:.3f}  |  PR-AUC: {prauc:.3f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T06:57:45.033194Z","iopub.execute_input":"2026-05-19T06:57:45.033564Z","iopub.status.idle":"2026-05-19T08:22:04.618748Z","shell.execute_reply.started":"2026-05-19T06:57:45.033527Z","shell.execute_reply":"2026-05-19T08:22:04.617670Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 3, figsize=(18, 5))\naxes[0].plot(history[\"train_loss\"], label=\"Train\", linewidth=2)\naxes[0].plot(history[\"val_loss\"],   label=\"Val\",   linewidth=2)\naxes[0].set_title(\"Loss\", fontsize=14, fontweight=\"bold\"); axes[0].legend()\naxes[1].plot(history[\"val_f1\"], color=\"#e74c3c\", linewidth=2)\naxes[1].set_title(\"Validation F1\", fontsize=14, fontweight=\"bold\")\naxes[2].plot(history[\"val_prauc\"], color=\"#8e44ad\", linewidth=2)\naxes[2].set_title(\"Validation PR-AUC\", fontsize=14, fontweight=\"bold\")\n\n\nfor ax in axes: ax.set_xlabel(\"Epoch\")\nplt.tight_layout();\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T08:22:05.038175Z","iopub.execute_input":"2026-05-19T08:22:05.038494Z","iopub.status.idle":"2026-05-19T08:22:05.450416Z","shell.execute_reply.started":"2026-05-19T08:22:05.038459Z","shell.execute_reply":"2026-05-19T08:22:05.449932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"=\" * 60)\nprint(\"MODEL A — STATIC CNN (MobileNetV2)  ·  EVALUATION\")\nprint(\"=\" * 60)\nprint(classification_report(all_labels, preds, target_names=[\"Healthy\", \"Under-Extrusion\"]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T08:22:05.475771Z","iopub.execute_input":"2026-05-19T08:22:05.476190Z","iopub.status.idle":"2026-05-19T08:22:05.497287Z","shell.execute_reply.started":"2026-05-19T08:22:05.476140Z","shell.execute_reply":"2026-05-19T08:22:05.496396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(14, 5))\ncm = confusion_matrix(all_labels, preds)\n\nsns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Reds\", ax=axes[0],\n            xticklabels=[\"Healthy\", \"Defect\"], yticklabels=[\"Healthy\", \"Defect\"])\naxes[0].set_title(\"Confusion Matrix — Static CNN\", fontsize=14, fontweight=\"bold\")\naxes[0].set_ylabel(\"Actual\"); axes[0].set_xlabel(\"Predicted\")\n\nprec, rec, thresholds = precision_recall_curve(all_labels, probs)\n\naxes[1].plot(rec, prec, color=\"#e74c3c\", linewidth=2)\naxes[1].fill_between(rec, prec, alpha=0.15, color=\"#e74c3c\")\naxes[1].axhline(y=all_labels.mean(), color=\"gray\", linestyle=\"--\", label=f\"Baseline ({all_labels.mean():.3f})\")\naxes[1].set_title(f\"Precision-Recall Curve  ·  PR-AUC = {prauc:.3f}\", fontsize=14, fontweight=\"bold\")\naxes[1].set_xlabel(\"Recall\"); axes[1].set_ylabel(\"Precision\"); axes[1].legend()\n\nplt.tight_layout();\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T10:17:53.426666Z","iopub.execute_input":"2026-05-19T10:17:53.427008Z","iopub.status.idle":"2026-05-19T10:17:53.815188Z","shell.execute_reply.started":"2026-05-19T10:17:53.426974Z","shell.execute_reply":"2026-05-19T10:17:53.814439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cnn_probs  = probs.copy()\ncnn_labels = all_labels.copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T08:22:06.344556Z","iopub.execute_input":"2026-05-19T08:22:06.345173Z","iopub.status.idle":"2026-05-19T08:22:06.349225Z","shell.execute_reply.started":"2026-05-19T08:22:06.345146Z","shell.execute_reply":"2026-05-19T08:22:06.348302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Generating per-frame predictions for all validation prints...\")\n\nfull_val_ds = PrintDefectDataset(val_df, IMG_DIR, val_transforms)\nfull_val_loader = DataLoader(full_val_ds, batch_size=BATCH_SIZE, shuffle=False, num_workers=2)\nmodel_cnn.eval()\nframe_logits = []\n\nwith torch.no_grad():\n    for imgs, _ in full_val_loader:\n        imgs = imgs.to(DEVICE)\n        logits = model_cnn(imgs)\n        frame_logits.append(logits.cpu().numpy())\n\nframe_logits = np.concatenate(frame_logits)\nframe_probs  = 1 / (1 + np.exp(-frame_logits))\nval_scored = val_df.copy()\nval_scored[\"prob\"] = frame_probs\nval_scored[\"pred\"] = (frame_probs >= 0.5).astype(int)\n\n\nval_scored = val_scored.sort_values([\"print_id\", \"img_path\"]).reset_index(drop=True)\n\n\nval_scored[\"frame_idx\"] = val_scored.groupby(\"print_id\").cumcount()\nval_scored[\"total_frames\"] = val_scored.groupby(\"print_id\")[\"frame_idx\"].transform(\"max\") + 1\nval_scored[\"progress\"] = val_scored[\"frame_idx\"] / val_scored[\"total_frames\"]  # 0.0 → 1.0\n\nprint(f\"Scored {len(val_scored)} frames across {val_scored['print_id'].nunique()} print jobs.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T10:14:50.570747Z","iopub.execute_input":"2026-05-19T10:14:50.571490Z","iopub.status.idle":"2026-05-19T10:17:53.424804Z","shell.execute_reply.started":"2026-05-19T10:14:50.571460Z","shell.execute_reply":"2026-05-19T10:17:53.423916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"WINDOW = 8  # 4 seconds smoothing window \n\ndef compute_print_timeline(group):\n    \"\"\"For a single print job, compute smoothed probs and first alert.\"\"\"\n    probs = group[\"prob\"].values\n    n = len(probs)\n    \n    # Smoothed probability (rolling average)\n    if n >= WINDOW:\n        kernel = np.ones(WINDOW) / WINDOW\n        padded = np.pad(probs, (WINDOW - 1, 0), mode='edge')\n        smoothed = np.convolve(padded, kernel, mode='valid')\n    else:\n        smoothed = np.cumsum(probs) / np.arange(1, n + 1)\n        \n    # Trend: slope over sliding window\n    slopes = np.zeros(n)\n    for i in range(WINDOW, n):\n        seg = probs[i - WINDOW:i]\n        slopes[i] = np.polyfit(np.arange(WINDOW), seg, 1)[0]\n        \n    # First alert: first frame where smoothed prob exceeds threshold\n    alert_threshold = 0.5\n    alerts = np.where(smoothed >= alert_threshold)[0]\n    first_alert_frame = alerts[0] if len(alerts) > 0 else -1\n    first_alert_progress = first_alert_frame / n if first_alert_frame >= 0 else None\n    return smoothed, slopes, first_alert_frame, first_alert_progress\n\nprint_summaries = []\ntimeline_data = []\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T10:07:52.499514Z","iopub.execute_input":"2026-05-19T10:07:52.500387Z","iopub.status.idle":"2026-05-19T10:07:52.511703Z","shell.execute_reply.started":"2026-05-19T10:07:52.500357Z","shell.execute_reply":"2026-05-19T10:07:52.510934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for print_id, group in val_scored.groupby(\"print_id\"):\n    group = group.sort_values(\"frame_idx\").reset_index(drop=True)\n    true_label = group[\"has_under_extrusion\"].iloc[0]\n    smoothed, slopes, first_alert, alert_progress = compute_print_timeline(group)\n    for i in range(len(group)):\n        timeline_data.append({\n            \"print_id\": print_id,\n            \"frame_idx\": i,\n            \"progress\": group[\"progress\"].iloc[i],\n            \"true_label\": true_label,\n            \"raw_prob\": group[\"prob\"].iloc[i],\n            \"smoothed_prob\": smoothed[i],\n            \"trend_slope\": slopes[i],\n        })\n    print_summaries.append({\n        \"print_id\": print_id,\n        \"true_label\": true_label,\n        \"n_frames\": len(group),\n        \"mean_prob\": group[\"prob\"].mean(),\n        \"max_prob\": group[\"prob\"].max(),\n        \"smoothed_max\": smoothed.max(),\n        \"first_alert_frame\": first_alert,\n        \"first_alert_progress\": alert_progress,\n        \"detected\": first_alert >= 0,\n    })\ntimeline_df = pd.DataFrame(timeline_data)\nsummary_df  = pd.DataFrame(print_summaries)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T10:07:55.497403Z","iopub.execute_input":"2026-05-19T10:07:55.498060Z","iopub.status.idle":"2026-05-19T10:07:58.171247Z","shell.execute_reply.started":"2026-05-19T10:07:55.498031Z","shell.execute_reply":"2026-05-19T10:07:58.170506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"defective_prints = summary_df[summary_df[\"true_label\"] == 1]\nhealthy_prints   = summary_df[summary_df[\"true_label\"] == 0]\ndetected = defective_prints[defective_prints[\"detected\"]]\nmissed   = defective_prints[~defective_prints[\"detected\"]]\n\nprint(\"=\" * 70)\nprint(\"EARLY DETECTION ANALYSIS\")\nprint(\"=\" * 70)\nprint(f\"  Defective prints in validation : {len(defective_prints)}\")\nprint(f\"  Detected (alert triggered)     : {len(detected)}  ({len(detected)/max(len(defective_prints),1)*100:.0f}%)\")\nprint(f\"  Missed (no alert)              : {len(missed)}\")\nprint()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T10:08:15.318719Z","iopub.execute_input":"2026-05-19T10:08:15.319126Z","iopub.status.idle":"2026-05-19T10:08:15.328353Z","shell.execute_reply.started":"2026-05-19T10:08:15.319097Z","shell.execute_reply":"2026-05-19T10:08:15.327519Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if len(detected) > 0:\n    avg_alert = detected[\"first_alert_progress\"].mean() * 100\n    median_alert = detected[\"first_alert_progress\"].median() * 100\n    earliest = detected[\"first_alert_progress\"].min() * 100\n    print(f\"  Avg first alert at    : {avg_alert:.1f}% of print progress\")\n    print(f\"  Median first alert at : {median_alert:.1f}% of print progress\")\n    print(f\"  Earliest detection at : {earliest:.1f}% of print progress\")\n    print(f\"\\n On average, the system catches under-extrusion with\")\n    print(f\"    {100 - avg_alert:.0f}% of the print remaining  saving material and time from being wasted.\")\n\nfalse_alarms = healthy_prints[healthy_prints[\"detected\"]]\nprint(f\"\\n  False alarms on healthy prints : {len(false_alarms)}/{len(healthy_prints)}\")\nprint(f\"  False alarm rate              : {len(false_alarms)/max(len(healthy_prints),1)*100:.1f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T10:09:23.723476Z","iopub.execute_input":"2026-05-19T10:09:23.724279Z","iopub.status.idle":"2026-05-19T10:09:23.731205Z","shell.execute_reply.started":"2026-05-19T10:09:23.724248Z","shell.execute_reply":"2026-05-19T10:09:23.730374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(16, 6))\n# 15a — Distribution of first-alert timing\nif len(detected) > 0:\n    axes[0].hist(detected[\"first_alert_progress\"] * 100, bins=15,\n                 color=\"#e74c3c\", edgecolor=\"black\", alpha=0.8)\n    axes[0].axvline(x=detected[\"first_alert_progress\"].median() * 100,\n                    color=\"black\", linestyle=\"--\", linewidth=2,\n                    label=f'Median: {detected[\"first_alert_progress\"].median()*100:.0f}%')\n    axes[0].set_xlabel(\"Print Progress When First Alert Fires (%)\", fontsize=12)\n    axes[0].set_ylabel(\"Number of Prints\", fontsize=12)\n    axes[0].set_title(\"How Early Do We Catch Defects?\", fontsize=14, fontweight=\"bold\")\n    axes[0].legend(fontsize=11)\n    \nthresholds = np.arange(0.1, 0.95, 0.05)\ndetection_rates, false_alarm_rates = [], []\nfor t in thresholds:\n    det = 0\n    for _, row in defective_prints.iterrows():\n        pid = row[\"print_id\"]\n        tl = timeline_df[timeline_df[\"print_id\"] == pid]\n        if (tl[\"smoothed_prob\"] >= t).any():\n            det += 1\n    detection_rates.append(det / max(len(defective_prints), 1) * 100)\n    fa = 0\n    for _, row in healthy_prints.iterrows():\n        pid = row[\"print_id\"]\n        tl = timeline_df[timeline_df[\"print_id\"] == pid]\n        if (tl[\"smoothed_prob\"] >= t).any():\n            fa += 1\n    false_alarm_rates.append(fa / max(len(healthy_prints), 1) * 100)\naxes[1].plot(thresholds * 100, detection_rates, color=\"#e74c3c\",\n             linewidth=2.5, marker=\"o\", markersize=4, label=\"Defect Detection Rate\")\naxes[1].plot(thresholds * 100, false_alarm_rates, color=\"#3498db\",\n             linewidth=2.5, marker=\"s\", markersize=4, label=\"False Alarm Rate\")\naxes[1].set_xlabel(\"Alert Threshold (%)\", fontsize=12)\naxes[1].set_ylabel(\"Rate (%)\", fontsize=12)\naxes[1].set_title(\"The Operator's Trade-Off: Detection vs. False Alarms\",\n                   fontsize=14, fontweight=\"bold\")\naxes[1].legend(fontsize=11)\naxes[1].grid(True, alpha=0.3)\nplt.tight_layout();\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T10:10:45.643976Z","iopub.execute_input":"2026-05-19T10:10:45.644811Z","iopub.status.idle":"2026-05-19T10:10:46.241148Z","shell.execute_reply.started":"2026-05-19T10:10:45.644785Z","shell.execute_reply":"2026-05-19T10:10:46.240410Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Healthscore which can be shown as a dashboard while printing","metadata":{}},{"cell_type":"code","source":"def compute_health_score(smoothed_prob, trend_slope):\n    \"\"\"\n    Combine the current defect probability and the trend into a\n    single health score (0-100) that operators can monitor.\n      100 = perfect health\n        0 = critical, stop the print\n    \"\"\"\n    # Base score from probability (inverted: low prob = high health)\n    base = (1 - smoothed_prob) * 80\n    # Trend penalty: if quality is degrading, penalise further\n    trend_penalty = np.clip(trend_slope * 200, 0, 20)\n    score = np.clip(base - trend_penalty, 0, 100)\n    return score\ntimeline_df[\"health_score\"] = compute_health_score(\n    timeline_df[\"smoothed_prob\"].values,\n    timeline_df[\"trend_slope\"].values\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T10:11:29.035335Z","iopub.execute_input":"2026-05-19T10:11:29.036110Z","iopub.status.idle":"2026-05-19T10:11:29.042261Z","shell.execute_reply.started":"2026-05-19T10:11:29.036076Z","shell.execute_reply":"2026-05-19T10:11:29.041407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 2, figsize=(18, 10))\nfor row, label in enumerate([0, 1]):\n    sample_pids = timeline_df[timeline_df[\"true_label\"] == label][\"print_id\"].unique()[:2]\n    for col, pid in enumerate(sample_pids):\n        ax = axes[row, col]\n        pdata = timeline_df[timeline_df[\"print_id\"] == pid]\n        # Color the health score line by severity\n        health = pdata[\"health_score\"].values\n        progress = pdata[\"progress\"].values * 100\n        # Plot health score\n        ax.fill_between(progress, health, alpha=0.3,\n                        color=\"#2ecc71\" if label == 0 else \"#e74c3c\")\n        ax.plot(progress, health, linewidth=2.5,\n                color=\"#27ae60\" if label == 0 else \"#c0392b\")\n        # Alert zones\n        ax.axhspan(0, 30, alpha=0.08, color=\"red\", label=\"CRITICAL\")\n        ax.axhspan(30, 60, alpha=0.06, color=\"orange\", label=\"WARNING\")\n        ax.axhspan(60, 100, alpha=0.04, color=\"green\", label=\"HEALTHY\")\n        status = \"DEFECTIVE PRINT\" if label else \"HEALTHY PRINT\"\n        ax.set_title(f\"Print {pid} — {status}\", fontsize=13, fontweight=\"bold\")\n        ax.set_xlabel(\"Print Progress (%)\")\n        ax.set_ylabel(\"Health Score (0-100)\")\n        ax.set_ylim(-5, 105)\n        if row == 0 and col == 0:\n            ax.legend(loc=\"lower left\", fontsize=9)\nfig.suptitle(\"Operator Dashboard: Real-Time Print Health Score\",\n             fontsize=16, fontweight=\"bold\", y=1.02)\nplt.tight_layout(); plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T10:11:35.083577Z","iopub.execute_input":"2026-05-19T10:11:35.084387Z","iopub.status.idle":"2026-05-19T10:11:35.783792Z","shell.execute_reply.started":"2026-05-19T10:11:35.084356Z","shell.execute_reply":"2026-05-19T10:11:35.782998Z"}},"outputs":[],"execution_count":null},{"cell_type":"raw","source":"Alert System","metadata":{}},{"cell_type":"code","source":"def generate_alerts(print_id, timeline_df):\n    \"\"\"\n    Generate operator-facing alerts for a single print job.\n    Returns a list of (frame, severity, message) tuples.\n    \"\"\"\n    pdata = timeline_df[timeline_df[\"print_id\"] == print_id].sort_values(\"frame_idx\")\n    alerts = []\n    for _, row in pdata.iterrows():\n        h = row[\"health_score\"]\n        trend = row[\"trend_slope\"]\n        frame = int(row[\"frame_idx\"])\n        prog = row[\"progress\"] * 100\n        if h < 20:\n            alerts.append((frame, \"CRITICAL\",\n                f\"Frame {frame} ({prog:.0f}%) — Health score {h:.0f}/100. \"\n                f\"RECOMMEND: Pause print. Inspect extruder for clogging or \"\n                f\"filament feed issues.\"))\n        elif h < 40:\n            alerts.append((frame, \"WARNING\",\n                f\"Frame {frame} ({prog:.0f}%) — Health score {h:.0f}/100. \"\n                f\"Under-extrusion indicators rising. Check nozzle temperature \"\n                f\"and material flow rate.\"))\n        elif h < 60 and trend > 0.02:\n            alerts.append((frame, \"WATCH\",\n                f\"Frame {frame} ({prog:.0f}%) — Health declining (trend: \"\n                f\"{trend:.3f}). Quality is degrading. Monitor closely.\"))\n    return alerts\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T10:12:12.125789Z","iopub.execute_input":"2026-05-19T10:12:12.126572Z","iopub.status.idle":"2026-05-19T10:12:12.132675Z","shell.execute_reply.started":"2026-05-19T10:12:12.126528Z","shell.execute_reply":"2026-05-19T10:12:12.131634Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"=\" * 70)\nprint(\"AUTOMATED ALERT SYSTEM — SAMPLE OUTPUT\")\nprint(\"=\" * 70)\nfor label, label_name in [(1, \"DEFECTIVE\"), (0, \"HEALTHY\")]:\n    pid = timeline_df[timeline_df[\"true_label\"] == label][\"print_id\"].unique()[0]\n    alerts = generate_alerts(pid, timeline_df)\n    print(f\"\\n📋 Print {pid} (Ground Truth: {label_name})\")\n    print(\"-\" * 60)\n    if alerts:\n        # Show first 3 and last alert to avoid flooding output\n        shown = alerts[:3]\n        if len(alerts) > 4:\n            print(f\"  ... ({len(alerts) - 4} more alerts) ...\")\n            shown.append(alerts[-1])\n        elif len(alerts) == 4:\n            shown.append(alerts[3])\n        for _, severity, msg in shown:\n            print(f\"  {severity}: {msg}\")\n    else:\n        print(\"  No alerts — print is healthy throughout.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T10:12:25.437940Z","iopub.execute_input":"2026-05-19T10:12:25.438525Z","iopub.status.idle":"2026-05-19T10:12:25.560242Z","shell.execute_reply.started":"2026-05-19T10:12:25.438499Z","shell.execute_reply":"2026-05-19T10:12:25.559574Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Notebook Summary","metadata":{}},{"cell_type":"code","source":"print(\"\\n\" + \"=\" * 70)\nprint(\"PRINT-LEVEL CLASSIFICATION (Job-Level Decision)\")\nprint(\"=\" * 70)\nprint(\"\"\"\nIn production, we aggregate frame predictions\ninto a single per-job decision to determine if the print will fail.\n\"\"\")\nsummary_df[\"job_pred_mean\"]  = (summary_df[\"mean_prob\"] >= 0.5).astype(int)\nsummary_df[\"job_pred_max\"]   = (summary_df[\"max_prob\"] >= 0.5).astype(int)\nsummary_df[\"job_pred_smooth\"] = (summary_df[\"smoothed_max\"] >= 0.5).astype(int)\njob_true = summary_df[\"true_label\"].values\nfor method, col in [(\"Mean Probability\", \"job_pred_mean\"),\n                     (\"Max Probability\",  \"job_pred_max\"),\n                     (\"Smoothed Max\",     \"job_pred_smooth\")]:\n    job_pred = summary_df[col].values\n    f1 = f1_score(job_true, job_pred)\n    cm = confusion_matrix(job_true, job_pred)\n    tn, fp, fn, tp = cm.ravel()\n    print(f\"  {method:20s}  |  F1: {f1:.3f}  |  TP: {tp}  FP: {fp}  FN: {fn}  TN: {tn}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-19T10:13:20.051272Z","iopub.execute_input":"2026-05-19T10:13:20.051851Z","iopub.status.idle":"2026-05-19T10:13:20.070654Z","shell.execute_reply.started":"2026-05-19T10:13:20.051821Z","shell.execute_reply":"2026-05-19T10:13:20.069755Z"}},"outputs":[],"execution_count":null}]}