{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":11848,"databundleVersionId":862157},{"sourceType":"datasetVersion","sourceId":11016738,"datasetId":6859474,"databundleVersionId":11398150},{"sourceType":"datasetVersion","sourceId":16060032,"datasetId":10299194,"databundleVersionId":17027902},{"sourceType":"datasetVersion","sourceId":16085767,"datasetId":10306282,"databundleVersionId":17055561},{"sourceType":"datasetVersion","sourceId":16061082,"datasetId":10299778,"databundleVersionId":17029036},{"sourceType":"kernelVersion","sourceId":12131586}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"b894f75a-0a70-4c04-81e5-804970068475","cell_type":"markdown","source":"# PCam Test Set — Feature Extraction & Model Evaluation\n\nThis notebook evaluates our trained model(s) on the **official PCam test split** with real ground-truth labels.","metadata":{}},{"id":"1f8534f9-f178-4ab5-abe1-cc6073e672aa","cell_type":"markdown","source":"---\n## 1. Setup & Dependencies","metadata":{}},{"id":"b7629f51-3033-41ed-af67-0d7efe236f21","cell_type":"code","source":"import subprocess, sys\n\nrequired = ['scikit-learn', 'scikit-image', 'pandas', 'numpy',\n            'matplotlib', 'seaborn', 'joblib', 'scipy', 'h5py', 'opencv-python', 'tqdm']\nfor pkg in required:\n    try:\n        __import__(pkg.replace('-', '_').replace('opencv_python', 'cv2'))\n        print(f'  OK: {pkg}')\n    except ImportError:\n        print(f'  Installing: {pkg}')\n        subprocess.check_call([sys.executable, '-m', 'pip', 'install', pkg, '-q'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T17:02:02.525340Z","iopub.execute_input":"2026-05-07T17:02:02.525642Z","iopub.status.idle":"2026-05-07T17:02:14.523030Z","shell.execute_reply.started":"2026-05-07T17:02:02.525600Z","shell.execute_reply":"2026-05-07T17:02:14.522015Z"}},"outputs":[],"execution_count":null},{"id":"69181c8f-da96-456a-b852-4bd1dbaa3a71","cell_type":"code","source":"import h5py\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport joblib\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm import tqdm\nfrom sklearn.metrics import (\n    roc_auc_score, roc_curve, accuracy_score,\n    classification_report, confusion_matrix, ConfusionMatrixDisplay\n)\nimport warnings\nwarnings.filterwarnings('ignore')\n\nsns.set_theme(style='whitegrid', palette='muted')\nplt.rcParams.update({'figure.dpi': 120, 'font.size': 11})\nprint(f'numpy {np.__version__} | h5py {h5py.__version__}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T17:03:55.652125Z","iopub.execute_input":"2026-05-07T17:03:55.652512Z","iopub.status.idle":"2026-05-07T17:03:55.829091Z","shell.execute_reply.started":"2026-05-07T17:03:55.652478Z","shell.execute_reply":"2026-05-07T17:03:55.828145Z"}},"outputs":[],"execution_count":null},{"id":"647b9b82-2840-4844-bb2b-138a48de0a40","cell_type":"markdown","source":"---\n## 2. Clone Feature Extraction Repo","metadata":{}},{"id":"1e0803dc-5563-4288-b658-7e4c284c5a0a","cell_type":"code","source":"!git clone https://github.com/mostafaayman646/Histopathologic-Cancer-Detection.git /kaggle/working/Histopathologic-Cancer-Detection\n\n# Add repo to path so we can import from it\nimport sys\nsys.path.insert(0, '/kaggle/working/Histopathologic-Cancer-Detection')\nprint('Repo cloned and added to path.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T17:04:03.640651Z","iopub.execute_input":"2026-05-07T17:04:03.641153Z","iopub.status.idle":"2026-05-07T17:04:04.695012Z","shell.execute_reply.started":"2026-05-07T17:04:03.641122Z","shell.execute_reply":"2026-05-07T17:04:04.693694Z"}},"outputs":[],"execution_count":null},{"id":"e8c6d406-0eaf-4c95-9584-9ed540b821f8","cell_type":"markdown","source":"---\n## 3. Configuration — Set Your Paths Here","metadata":{}},{"id":"46986628-401a-41d2-9b93-1a4c5fb93667","cell_type":"code","source":"# ── EDIT THESE PATHS ────────────────────────────────────────────────────────\nX_H5_PATH    = '/kaggle/input/datasets/rajprog/camelyonpatch-train-valid-test/camelyonpatch_level_2_split_test_x.h5'\nY_H5_PATH    = '/kaggle/input/datasets/rajprog/camelyonpatch-train-valid-test/camelyonpatch_level_2_split_test_y.h5'\nMODEL_PATH   = '/kaggle/input/datasets/marsh04/pkl-models/pipeline_StandardScaler_PCA40_SVM.pkl'\nOUTPUT_NPZ   = '/kaggle/working/pcam_test_features.npz'   # where extracted features are saved\n\nFEATURES     = ['dog', 'color', 'glcm', 'lbp', 'lbglcm', 'glrlm', 'sfta']\nN_SAMPLES    = -1   # -1 = all (~32768 in PCam test split)\n# ────────────────────────────────────────────────────────────────────────────\n\nprint('Config set.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T17:04:27.876429Z","iopub.execute_input":"2026-05-07T17:04:27.876872Z","iopub.status.idle":"2026-05-07T17:04:27.884374Z","shell.execute_reply.started":"2026-05-07T17:04:27.876831Z","shell.execute_reply":"2026-05-07T17:04:27.883506Z"}},"outputs":[],"execution_count":null},{"id":"be2f07be-ee56-4d7b-b89e-396c238ff94d","cell_type":"markdown","source":"---\n## 4. Inspect H5 Files","metadata":{}},{"id":"9abc0d3f-9e2b-4c26-923b-1595380805db","cell_type":"code","source":"with h5py.File(X_H5_PATH, 'r') as fx:\n    x_key   = list(fx.keys())[0]\n    x_shape = fx[x_key].shape\n    x_dtype = fx[x_key].dtype\n    print(f'Images key   : \"{x_key}\"')\n    print(f'Images shape : {x_shape}   (N x H x W x C)')\n    print(f'Images dtype : {x_dtype}')\n\nwith h5py.File(Y_H5_PATH, 'r') as fy:\n    y_key   = list(fy.keys())[0]\n    y_shape = fy[y_key].shape\n    y_vals  = fy[y_key][:].flatten()\n    print(f'\\nLabels key   : \"{y_key}\"')\n    print(f'Labels shape : {y_shape}')\n    print(f'Unique labels: {np.unique(y_vals)}  |  Cancer: {y_vals.sum():,}  Healthy: {(y_vals==0).sum():,}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T17:04:33.390271Z","iopub.execute_input":"2026-05-07T17:04:33.390661Z","iopub.status.idle":"2026-05-07T17:04:33.432774Z","shell.execute_reply.started":"2026-05-07T17:04:33.390616Z","shell.execute_reply":"2026-05-07T17:04:33.431587Z"}},"outputs":[],"execution_count":null},{"id":"07dff1d4-84b0-4d24-b754-7cd285fdb8ba","cell_type":"markdown","source":"---\n## 5. Feature Extraction from H5","metadata":{}},{"id":"b448e037-f108-470d-af8d-23d4a0fd48f3","cell_type":"code","source":"from skimage.feature import blob_dog\nimport image_features as feat_lib\n\ndef extract_features(img_bgr, feature_list):\n    \"\"\"Extract and concatenate all requested feature sets from a BGR image.\n    Mirrors the logic in DataHandler.build_dataset() exactly.\"\"\"\n    img_gray = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2GRAY)\n    parts = []\n    if 'dog' in feature_list:\n        parts.append(feat_lib.extract_advanced_stats(\n            img_gray, blob_dog(img_gray, min_sigma=2, max_sigma=15, threshold=0.1)))\n    if 'color' in feature_list:\n        parts.append(feat_lib.extract_color_stats(img_bgr))\n    if 'glcm' in feature_list:\n        parts.append(feat_lib.extract_glcm_features(img_gray))\n    if 'lbp' in feature_list:\n        parts.append(feat_lib.extract_lbp_features(img_gray))\n    if 'lbglcm' in feature_list:\n        parts.append(feat_lib.extract_lbglcm_features(img_gray))\n    if 'glrlm' in feature_list:\n        parts.append(feat_lib.extract_glrlm_features(img_gray))\n    if 'sfta' in feature_list:\n        parts.append(feat_lib.extract_sfta_features(img_gray, n_levels=3))\n    return np.concatenate(parts)\n\nprint('extract_features helper defined using image_features module.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T17:12:44.931872Z","iopub.execute_input":"2026-05-07T17:12:44.932215Z","iopub.status.idle":"2026-05-07T17:12:44.943636Z","shell.execute_reply.started":"2026-05-07T17:12:44.932178Z","shell.execute_reply":"2026-05-07T17:12:44.942597Z"}},"outputs":[],"execution_count":null},{"id":"3b5cac51-2e5a-4365-afd1-27f098b2765a","cell_type":"code","source":"import os\n\n# Load from cache if already extracted\nif os.path.exists(OUTPUT_NPZ):\n    print(f'Found cached features at {OUTPUT_NPZ} — loading...')\n    data   = np.load(OUTPUT_NPZ)\n    X_pcam = data['X'].astype(np.float32)\n    y_true = data['y'].astype(int)\n    print(f'Loaded: {X_pcam.shape[0]:,} samples x {X_pcam.shape[1]} features')\nelse:\n    print('No cache found — extracting features from H5...')\n\n    with h5py.File(X_H5_PATH, 'r') as fx, h5py.File(Y_H5_PATH, 'r') as fy:\n        x_key = list(fx.keys())[0]\n        y_key = list(fy.keys())[0]\n\n        total     = fx[x_key].shape[0]\n        n_extract = total if N_SAMPLES == -1 else min(N_SAMPLES, total)\n\n        y_true = fy[y_key][:n_extract].flatten().astype(int)\n\n        print(f'Extracting features from {n_extract:,} images...')\n        features_list = []\n        failed        = []\n\n        for i in tqdm(range(n_extract), desc='Extracting'):\n            try:\n                # PCam stores images as RGB uint8 — convert to BGR for OpenCV consistency\n                img_rgb = fx[x_key][i]           # (96, 96, 3) uint8\n                img_bgr = cv2.cvtColor(img_rgb, cv2.COLOR_RGB2BGR)\n\n                feat = extract_features(img_bgr, FEATURES)\n                features_list.append(feat)\n            except Exception as e:\n                failed.append((i, str(e)))\n                features_list.append(np.zeros_like(features_list[0]) if features_list else None)\n\n        if failed:\n            print(f'WARNING: {len(failed)} images failed extraction.')\n            for idx, err in failed[:5]:\n                print(f'  Sample {idx}: {err}')\n\n        X_pcam = np.array([f for f in features_list if f is not None], dtype=np.float32)\n\n        np.savez_compressed(OUTPUT_NPZ, X=X_pcam, y=y_true)\n        print(f'\\nSaved to {OUTPUT_NPZ}')\n        print(f'Shape: {X_pcam.shape[0]:,} samples x {X_pcam.shape[1]} features')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T17:12:52.399176Z","iopub.execute_input":"2026-05-07T17:12:52.399552Z","iopub.status.idle":"2026-05-07T17:26:46.267577Z","shell.execute_reply.started":"2026-05-07T17:12:52.399523Z","shell.execute_reply":"2026-05-07T17:26:46.266622Z"}},"outputs":[],"execution_count":null},{"id":"1302ce4d-79d4-4cc6-9c1f-e00dd7983bc3","cell_type":"markdown","source":"---\n## 6. Sanity Check — Feature Dimensions","metadata":{}},{"id":"eeae3d59-8e80-4d0e-9ba3-5ce18eb92f44","cell_type":"code","source":"EXPECTED_FEATURES = 51  # must match training\n\nn_samples, n_features = X_pcam.shape\nprint(f'Samples  : {n_samples:,}')\nprint(f'Features : {n_features}  (expected {EXPECTED_FEATURES})')\n\nassert n_features == EXPECTED_FEATURES, (\n    f'Feature dimension mismatch! Got {n_features}, expected {EXPECTED_FEATURES}. '\n    f'Check that FEATURES list matches exactly what was used during training.'\n)\n\nprint(f'\\nLabel distribution:')\nprint(f'  Cancer  (1): {y_true.sum():,}  ({y_true.mean()*100:.1f}%)')\nprint(f'  Healthy (0): {(y_true==0).sum():,}  ({(1-y_true.mean())*100:.1f}%)')\nprint(f'\\nFeature stats:')\nprint(f'  NaN count : {np.isnan(X_pcam).sum()}')\nprint(f'  Inf count : {np.isinf(X_pcam).sum()}')\nprint(f'  Min/Max   : {X_pcam.min():.4f} / {X_pcam.max():.4f}')\n\n# Replace any NaN/Inf with 0 (same safe fallback as training)\nX_pcam = np.nan_to_num(X_pcam, nan=0.0, posinf=0.0, neginf=0.0)\nprint('NaN/Inf replaced with 0.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T17:33:28.227115Z","iopub.execute_input":"2026-05-07T17:33:28.228139Z","iopub.status.idle":"2026-05-07T17:33:28.262147Z","shell.execute_reply.started":"2026-05-07T17:33:28.228099Z","shell.execute_reply":"2026-05-07T17:33:28.261012Z"}},"outputs":[],"execution_count":null},{"id":"6cfc13fa-f5aa-4257-a569-bb0e2dedca75","cell_type":"markdown","source":"---\n## 7. Load Model & Run Inference","metadata":{}},{"id":"57ebeb54-2bf1-4489-aa68-822d9c79d82f","cell_type":"code","source":"pipeline   = joblib.load(MODEL_PATH)\nmodel_step = pipeline.steps[-1][1]\n\nprint(f'Pipeline steps : {[s[0] for s in pipeline.steps]}')\nprint(f'Estimator      : {type(model_step).__name__}')\n\n# Verify feature count against pipeline's first step\nfirst_step = pipeline.steps[0][1]\nn_expected = getattr(first_step, 'n_features_in_', None)\nif n_expected:\n    assert X_pcam.shape[1] == n_expected, \\\n        f'Mismatch: data has {X_pcam.shape[1]} features, pipeline expects {n_expected}'\n    print(f'Feature count verified: {n_expected}')\nelse:\n    print(f'Cannot auto-verify feature count for {type(first_step).__name__} — proceeding.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T17:33:43.388765Z","iopub.execute_input":"2026-05-07T17:33:43.389122Z","iopub.status.idle":"2026-05-07T17:33:43.982151Z","shell.execute_reply.started":"2026-05-07T17:33:43.389092Z","shell.execute_reply":"2026-05-07T17:33:43.980954Z"}},"outputs":[],"execution_count":null},{"id":"147667f5-4aa8-4c5b-ad17-e237c9f231ea","cell_type":"code","source":"if hasattr(model_step, 'predict_proba'):\n    y_score = pipeline.predict_proba(X_pcam)[:, 1]\n    print('Using predict_proba for probability scores.')\nelif hasattr(model_step, 'decision_function'):\n    raw = pipeline.decision_function(X_pcam)\n    # Normalise decision scores to [0, 1] for a consistent threshold\n    y_score = (raw - raw.min()) / (raw.max() - raw.min() + 1e-9)\n    print('Using decision_function (normalised) for scores — predict_proba not available.')\nelse:\n    raise RuntimeError(\n        f'{type(model_step).__name__} supports neither predict_proba nor decision_function. '\n        'AUC-ROC cannot be computed.')\n\nTHRESHOLD = 0.5\ny_pred    = (y_score >= THRESHOLD).astype(int)\n\nprint(f'Inference complete.')\nprint(f'Score range : [{y_score.min():.4f}, {y_score.max():.4f}]')\nprint(f'Score mean  : {y_score.mean():.4f}')\nprint(f'Predicted cancer  : {y_pred.sum():,}  ({y_pred.mean()*100:.1f}%)')\nprint(f'Predicted healthy : {(y_pred==0).sum():,}  ({(1-y_pred.mean())*100:.1f}%)')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T17:33:51.577131Z","iopub.execute_input":"2026-05-07T17:33:51.577650Z","iopub.status.idle":"2026-05-07T17:37:48.008396Z","shell.execute_reply.started":"2026-05-07T17:33:51.577621Z","shell.execute_reply":"2026-05-07T17:37:48.007508Z"}},"outputs":[],"execution_count":null},{"id":"cdc039bf-3b0e-4641-b2d8-59533bc3c993","cell_type":"markdown","source":"---\n## 8. Evaluation Against Real Labels","metadata":{}},{"id":"1b5362d7-7623-4b6a-beac-bb373d24c811","cell_type":"code","source":"auc      = roc_auc_score(y_true, y_score)\nacc      = accuracy_score(y_true, y_pred)\nreport   = classification_report(y_true, y_pred, target_names=['Healthy', 'Cancer'])\n\nprint('=' * 50)\nprint('       EVALUATION ON PCAM TEST SET (REAL LABELS)')\nprint('=' * 50)\nprint(f'  AUC-ROC  : {auc:.4f}')\nprint(f'  Accuracy : {acc*100:.2f}%')\nprint()\nprint(report)\nprint('=' * 50)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T17:38:14.791446Z","iopub.execute_input":"2026-05-07T17:38:14.791807Z","iopub.status.idle":"2026-05-07T17:38:14.834099Z","shell.execute_reply.started":"2026-05-07T17:38:14.791778Z","shell.execute_reply":"2026-05-07T17:38:14.832458Z"}},"outputs":[],"execution_count":null},{"id":"d1de358a-5fe2-41d2-aab8-c5c6279f514c","cell_type":"markdown","source":"---\n## 9. Visualizations","metadata":{}},{"id":"f768bd69-18e4-43d6-9892-09121b2e4487","cell_type":"code","source":"# ── A. Score Distributions by True Class ────────────────────────────────────\nfig, axes = plt.subplots(1, 2, figsize=(14, 5))\nfig.suptitle('Score Distribution on PCam Test Set', fontsize=14, fontweight='bold')\n\nfor cls, label, color in [(0, 'Healthy', '#59A14F'), (1, 'Cancer', '#E15759')]:\n    axes[0].hist(y_score[y_true == cls], bins=60, alpha=0.7, color=color, label=label, edgecolor='white')\naxes[0].axvline(THRESHOLD, color='black', lw=2, linestyle='--', label='Threshold=0.5')\naxes[0].set_title('Score by True Class', fontweight='bold')\naxes[0].set_xlabel('Predicted Score'); axes[0].set_ylabel('Count')\naxes[0].legend()\n\n# Overall distribution\naxes[1].hist(y_score, bins=60, color='#4E79A7', edgecolor='white', alpha=0.85)\naxes[1].axvline(THRESHOLD, color='crimson', lw=2, linestyle='--', label='Threshold=0.5')\naxes[1].axvline(y_score.mean(), color='orange', lw=2, linestyle='-.', label=f'Mean={y_score.mean():.3f}')\naxes[1].set_title('Overall Score Distribution', fontweight='bold')\naxes[1].set_xlabel('Predicted Score'); axes[1].set_ylabel('Count')\naxes[1].legend()\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T17:38:25.283132Z","iopub.execute_input":"2026-05-07T17:38:25.283506Z","iopub.status.idle":"2026-05-07T17:38:26.073616Z","shell.execute_reply.started":"2026-05-07T17:38:25.283475Z","shell.execute_reply":"2026-05-07T17:38:26.072647Z"}},"outputs":[],"execution_count":null},{"id":"4ed52863-63a1-4a93-8891-7ac973d7bf63","cell_type":"code","source":"# ── B. ROC Curve ─────────────────────────────────────────────────────────────\nfpr, tpr, _ = roc_curve(y_true, y_score)\n\nfig, axes = plt.subplots(1, 2, figsize=(14, 5))\nfig.suptitle('ROC Curve & Confusion Matrix — PCam Test Set', fontsize=14, fontweight='bold')\n\naxes[0].plot(fpr, tpr, color='#4E79A7', lw=2.5, label=f'AUC = {auc:.4f}')\naxes[0].plot([0,1],[0,1], 'k--', lw=1.5, alpha=0.5, label='Random')\naxes[0].fill_between(fpr, tpr, alpha=0.08, color='#4E79A7')\naxes[0].set_title(f'ROC Curve (AUC = {auc:.4f})', fontweight='bold')\naxes[0].set_xlabel('False Positive Rate'); axes[0].set_ylabel('True Positive Rate')\naxes[0].legend(fontsize=10)\n\n# ── C. Confusion Matrix ───────────────────────────────────────────────────────\ncm = confusion_matrix(y_true, y_pred)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=['Healthy', 'Cancer'])\ndisp.plot(ax=axes[1], colorbar=True, cmap='Blues')\naxes[1].set_title(f'Confusion Matrix (Acc = {acc*100:.2f}%)', fontweight='bold')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T17:38:34.672453Z","iopub.execute_input":"2026-05-07T17:38:34.672825Z","iopub.status.idle":"2026-05-07T17:38:35.162311Z","shell.execute_reply.started":"2026-05-07T17:38:34.672793Z","shell.execute_reply":"2026-05-07T17:38:35.161011Z"}},"outputs":[],"execution_count":null},{"id":"9d128eec-4f1b-4d2f-bf65-4407d47784b0","cell_type":"code","source":"# ── D. CDF Comparison ─────────────────────────────────────────────────────────\nfig, ax = plt.subplots(figsize=(8, 5))\n\nfor cls, label, color in [(0, 'Healthy (true)', '#59A14F'), (1, 'Cancer (true)', '#E15759')]:\n    scores_cls = np.sort(y_score[y_true == cls])\n    cdf_cls    = np.linspace(0, 1, len(scores_cls))\n    ax.plot(scores_cls, cdf_cls, color=color, lw=2.5, label=label)\n\nax.axvline(THRESHOLD, color='black', lw=1.5, linestyle='--', label='Threshold=0.5')\nax.set_title('CDF of Scores by True Class', fontsize=13, fontweight='bold')\nax.set_xlabel('Predicted Score'); ax.set_ylabel('Cumulative Fraction')\nax.legend(fontsize=10)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T17:38:44.342384Z","iopub.execute_input":"2026-05-07T17:38:44.342727Z","iopub.status.idle":"2026-05-07T17:38:44.615126Z","shell.execute_reply.started":"2026-05-07T17:38:44.342689Z","shell.execute_reply":"2026-05-07T17:38:44.614322Z"}},"outputs":[],"execution_count":null},{"id":"d203a23d-3729-4a45-ba0d-9e7ed97e111b","cell_type":"code","source":"# ── E. Predicted vs Actual Class Distribution ───────────────────────────────────\nactual_healthy  = (y_true == 0).sum()\nactual_cancer   = (y_true == 1).sum()\npred_healthy    = (y_pred == 0).sum()\npred_cancer     = (y_pred == 1).sum()\n\ncategories = ['Healthy', 'Cancerous']\nactual_counts = [actual_healthy, actual_cancer]\npred_counts   = [pred_healthy,   pred_cancer]\n\nx = np.arange(len(categories))\nwidth = 0.35\n\nfig, ax = plt.subplots(figsize=(8, 5))\nbars1 = ax.bar(x - width/2, actual_counts, width, label='Actual',    color='steelblue',  alpha=0.85)\nbars2 = ax.bar(x + width/2, pred_counts,   width, label='Predicted', color='darkorange', alpha=0.85)\n\nax.set_title('Predicted vs Actual Class Distribution', fontsize=13, fontweight='bold')\nax.set_ylabel('Number of Samples')\nax.set_xticks(x)\nax.set_xticklabels(categories)\nax.legend()\n\nfor bar in bars1 + bars2:\n    ax.text(bar.get_x() + bar.get_width() / 2, bar.get_height() + 150,\n            f'{int(bar.get_height()):,}', ha='center', va='bottom', fontsize=9)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T17:42:57.185407Z","iopub.execute_input":"2026-05-07T17:42:57.185754Z","iopub.status.idle":"2026-05-07T17:42:57.414045Z","shell.execute_reply.started":"2026-05-07T17:42:57.185720Z","shell.execute_reply":"2026-05-07T17:42:57.412694Z"}},"outputs":[],"execution_count":null},{"id":"f421b4c2-42e9-4867-b225-21957c4884a8","cell_type":"markdown","source":"---\n## 10. Final Summary","metadata":{}},{"id":"ddae991e-3f00-4e83-ae10-ff57eaa49039","cell_type":"code","source":"cm = confusion_matrix(y_true, y_pred)\ntn, fp, fn, tp = cm.ravel()\n\nsensitivity = tp / (tp + fn)  # recall for cancer class\nspecificity  = tn / (tn + fp)\nppv          = tp / (tp + fp)  # precision for cancer class\nnpv          = tn / (tn + fn)\n\nn_samples  = X_pcam.shape[0]\nn_features = X_pcam.shape[1]\n\nprint('=' * 55)\nprint('       MODEL EVALUATION — PCam TEST SET (REAL LABELS)')\nprint('=' * 55)\nprint(f'  Test samples         : {n_samples:,}')\nprint(f'  Features             : {n_features}')\nprint()\nprint(f'  AUC-ROC              : {auc:.4f}')\nprint(f'  Accuracy             : {acc*100:.2f}%')\nprint()\nprint(f'  Sensitivity (Recall) : {sensitivity*100:.2f}%  (cancer correctly detected)')\nprint(f'  Specificity          : {specificity*100:.2f}%  (healthy correctly excluded)')\nprint(f'  Precision (PPV)      : {ppv*100:.2f}%')\nprint(f'  NPV                  : {npv*100:.2f}%')\nprint()\nprint(f'  True  Positives      : {tp:,}')\nprint(f'  True  Negatives      : {tn:,}')\nprint(f'  False Positives      : {fp:,}  (healthy called cancer)')\nprint(f'  False Negatives      : {fn:,}  (cancer missed)')\nprint('=' * 55)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T17:38:51.265705Z","iopub.execute_input":"2026-05-07T17:38:51.266085Z","iopub.status.idle":"2026-05-07T17:38:51.278760Z","shell.execute_reply.started":"2026-05-07T17:38:51.266053Z","shell.execute_reply":"2026-05-07T17:38:51.277474Z"}},"outputs":[],"execution_count":null}]}