{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":11848,"databundleVersionId":862157,"isSourceIdPinned":false}],"dockerImageVersionId":31329,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":21775.728078,"end_time":"2026-05-06T06:29:44.458862+00:00","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-05-06T00:26:48.730784+00:00","version":"2.7.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"c4b7bbed","cell_type":"code","source":"import os\nimport random\nimport shutil\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nimport math\nfrom IPython.display import display, HTML, Markdown\n\n# Visualizations\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nfrom matplotlib.patches import Patch, Rectangle\n\n# Image Processing\nimport cv2\nfrom skimage import io\nfrom skimage.transform import rotate\nfrom tifffile import imread\n\n# TensorFlow\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models, optimizers\nfrom tensorflow.keras.utils import image_dataset_from_directory\n\n# Keras\nfrom tensorflow.keras.layers import (\n    RandomFlip, RandomRotation, RandomZoom,\n    Conv2D, MaxPooling2D, AveragePooling2D,\n    Flatten, Dense, Dropout, BatchNormalization\n)\n\n# Profiling\nimport pandas_profiling as pp\n\n# Scikit-learn\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score, accuracy_score\n\nimport os\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, roc_auc_score\nfrom skimage import io\nfrom tensorflow.keras import layers, models, callbacks, optimizers\n\nfrom sklearn.model_selection import StratifiedKFold","metadata":{"execution":{"iopub.status.busy":"2026-05-12T15:57:58.547561Z","iopub.execute_input":"2026-05-12T15:57:58.548213Z","iopub.status.idle":"2026-05-12T15:57:58.554696Z","shell.execute_reply.started":"2026-05-12T15:57:58.548182Z","shell.execute_reply":"2026-05-12T15:57:58.553879Z"},"papermill":{"duration":35.800326,"end_time":"2026-05-06T00:35:18.347579+00:00","exception":false,"start_time":"2026-05-06T00:34:42.547253+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"dd1ff4d2","cell_type":"code","source":"# Load files\nsample_submission = pd.read_csv(\"/kaggle/input/competitions/histopathologic-cancer-detection/sample_submission.csv\")\ntrain_raw = pd.read_csv(\"/kaggle/input/competitions/histopathologic-cancer-detection/train_labels.csv\")\n\ntrain_path = \"/kaggle/input/competitions/histopathologic-cancer-detection/train/\"\ntest_path = \"/kaggle/input/competitions/histopathologic-cancer-detection/test/\"","metadata":{"execution":{"iopub.status.busy":"2026-05-12T15:57:58.560718Z","iopub.execute_input":"2026-05-12T15:57:58.561040Z","iopub.status.idle":"2026-05-12T15:57:58.808433Z","shell.execute_reply.started":"2026-05-12T15:57:58.561017Z","shell.execute_reply":"2026-05-12T15:57:58.807660Z"},"papermill":{"duration":1.028212,"end_time":"2026-05-06T00:35:19.812684+00:00","exception":false,"start_time":"2026-05-06T00:35:18.784472+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"34e9eb87","cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom PIL import Image\n\n# Visualization\nsample_id = train_raw.iloc[0]['id']\nimg = Image.open(train_path + sample_id + \".tif\")\n\nplt.imshow(img)\nplt.title(f\"Label: {train_raw.iloc[0]['label']}\")\nplt.show()\n\nprint(f\"Image Size: {img.size}\") # Should be 96x96","metadata":{"execution":{"iopub.status.busy":"2026-05-12T15:57:58.809649Z","iopub.execute_input":"2026-05-12T15:57:58.809941Z","iopub.status.idle":"2026-05-12T15:57:58.908085Z","shell.execute_reply.started":"2026-05-12T15:57:58.809916Z","shell.execute_reply":"2026-05-12T15:57:58.907360Z"},"papermill":{"duration":0.809938,"end_time":"2026-05-06T00:35:21.051664+00:00","exception":false,"start_time":"2026-05-06T00:35:20.241726+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"3c5e15ac","cell_type":"code","source":"# FILE PATHS\n\ntrain_files = [os.path.join(train_path, f\"{fid}.tif\") for fid in train_raw.id]\ntrain_targets = train_raw.label.values\n\ntest_files = sorted([\n    os.path.join(test_path, f) for f in os.listdir(test_path) if f.endswith(\".tif\")\n])","metadata":{"execution":{"iopub.status.busy":"2026-05-12T15:57:58.909083Z","iopub.execute_input":"2026-05-12T15:57:58.909411Z","iopub.status.idle":"2026-05-12T15:57:59.214475Z","shell.execute_reply.started":"2026-05-12T15:57:58.909376Z","shell.execute_reply":"2026-05-12T15:57:59.213568Z"},"papermill":{"duration":0.751269,"end_time":"2026-05-06T00:35:22.245655+00:00","exception":false,"start_time":"2026-05-06T00:35:21.494386+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"56fb2537","cell_type":"markdown","source":"## Model Architecture","metadata":{"papermill":{"duration":0.447201,"end_time":"2026-05-06T00:35:23.135906+00:00","exception":false,"start_time":"2026-05-06T00:35:22.688705+00:00","status":"completed"},"tags":[]}},{"id":"bb420045","cell_type":"markdown","source":"## FORTH CONFIGURATION ","metadata":{"papermill":{"duration":0.440385,"end_time":"2026-05-06T00:35:28.550996+00:00","exception":false,"start_time":"2026-05-06T00:35:28.110611+00:00","status":"completed"},"tags":[]}},{"id":"7f815911","cell_type":"code","source":"SEED = 42\nIMG_SIZE = (96, 96)\nBATCH_SIZE = 32\nEPOCHS = 20\nAUTOTUNE = tf.data.AUTOTUNE\n\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)","metadata":{"execution":{"iopub.status.busy":"2026-05-12T15:57:59.216362Z","iopub.execute_input":"2026-05-12T15:57:59.216669Z","iopub.status.idle":"2026-05-12T15:57:59.334802Z","shell.execute_reply.started":"2026-05-12T15:57:59.216645Z","shell.execute_reply":"2026-05-12T15:57:59.333861Z"},"papermill":{"duration":0.47597,"end_time":"2026-05-06T00:35:29.485288+00:00","exception":false,"start_time":"2026-05-06T00:35:29.009318+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"1577dff5","cell_type":"code","source":"files_arr  = np.array([f\"{train_path}{fid}.tif\" for fid in train_raw['id']])\nlabels_arr = np.array(train_raw['label'].values)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-12T15:57:59.336878Z","iopub.execute_input":"2026-05-12T15:57:59.337713Z","iopub.status.idle":"2026-05-12T15:57:59.513718Z","shell.execute_reply.started":"2026-05-12T15:57:59.337615Z","shell.execute_reply":"2026-05-12T15:57:59.512756Z"}},"outputs":[],"execution_count":null},{"id":"b2b8e893-1313-4b28-ad53-2a7612d93120","cell_type":"code","source":"print(len(files_arr))\nprint(len(labels_arr))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-12T15:57:59.514678Z","iopub.execute_input":"2026-05-12T15:57:59.515012Z","iopub.status.idle":"2026-05-12T15:57:59.519289Z","shell.execute_reply.started":"2026-05-12T15:57:59.514968Z","shell.execute_reply":"2026-05-12T15:57:59.518623Z"}},"outputs":[],"execution_count":null},{"id":"80c75d63","cell_type":"code","source":"METRICS = [\n    'accuracy',\n    tf.keras.metrics.AUC(name='roc_auc', curve='ROC')\n]\n\nCB = [\n    callbacks.EarlyStopping(monitor='val_roc_auc', patience=5, mode='max', restore_best_weights=True),\n    callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=2)\n]\n","metadata":{"execution":{"iopub.status.busy":"2026-05-12T15:57:59.520408Z","iopub.execute_input":"2026-05-12T15:57:59.520701Z","iopub.status.idle":"2026-05-12T15:57:59.547898Z","shell.execute_reply.started":"2026-05-12T15:57:59.520679Z","shell.execute_reply":"2026-05-12T15:57:59.547075Z"},"papermill":{"duration":2.11843,"end_time":"2026-05-06T00:35:34.329524+00:00","exception":false,"start_time":"2026-05-06T00:35:32.211094+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"7abe7618","cell_type":"code","source":"def compile_model(model, lr=1e-3):\n    model.compile(\n        optimizer=optimizers.Adam(learning_rate=lr),\n        loss='binary_crossentropy',\n        metrics=METRICS,\n    )\n    return model\n\n\ndef evaluate_model(model, dataset):\n    probs = model.predict(dataset, verbose=0).ravel()\n    labels = np.concatenate([y.numpy() for _, y in dataset], axis=0)\n    preds = (probs > 0.5).astype(int)\n    return {\n        'accuracy': accuracy_score(labels, preds),\n        'roc_auc': roc_auc_score(labels, probs)\n    }","metadata":{"execution":{"iopub.status.busy":"2026-05-12T15:57:59.548858Z","iopub.execute_input":"2026-05-12T15:57:59.549167Z","iopub.status.idle":"2026-05-12T15:57:59.554344Z","shell.execute_reply.started":"2026-05-12T15:57:59.549114Z","shell.execute_reply":"2026-05-12T15:57:59.553654Z"},"papermill":{"duration":0.499109,"end_time":"2026-05-06T00:35:35.283099+00:00","exception":false,"start_time":"2026-05-06T00:35:34.783990+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"e477305e","cell_type":"markdown","source":"#### COSINE LR Schedule with warmup","metadata":{}},{"id":"61b08379","cell_type":"code","source":"# Replaces ReduceLROnPlateau entirely.\n#\n# Why cosine instead of ReduceLROnPlateau:\n#   Your old config: factor=0.2, patience=2\n#   On a 200k dataset an epoch is ~1562 steps. By epoch 4-5 the plateau\n#   monitor often hasn't had enough time to reflect true convergence,\n#   so ReduceLROnPlateau fires and multiplies the LR by 0.2 — cutting it\n#   from 1e-4 to 2e-5 before the PASH prototypes have started moving.\n#   Cosine annealing is smooth and deterministic: the LR is never\n#   suddenly decimated, and the final low value lets the model fine-tune\n#   without overshooting.\n#\n# Why warmup:\n#   BDCG cross-attention and PASH prototypes both start random. Large\n#   early gradients can push prototypes to degenerate positions. A short\n#   warmup ramp prevents this without slowing overall training.\n# =============================================================================\n \ndef make_cosine_schedule(\n    total_epochs: int,\n    warmup_epochs: int,\n    base_lr: float\n) -> callbacks.LearningRateScheduler:\n    \"\"\"\n    Returns a LearningRateScheduler callback.\n \n    Phase 1 (epochs 0 → warmup_epochs):\n        Linear ramp from base_lr × 0.1  →  base_lr\n \n    Phase 2 (epochs warmup_epochs → total_epochs):\n        Cosine decay from base_lr  →  base_lr × 0.01\n    \"\"\"\n    def schedule(epoch: int, _current_lr: float) -> float:\n        if epoch < warmup_epochs:\n            # Linear warmup\n            return base_lr * 0.1 + base_lr * 0.9 * (epoch / max(1, warmup_epochs))\n        else:\n            # Cosine decay\n            progress = (epoch - warmup_epochs) / max(1, total_epochs - warmup_epochs)\n            cosine_val = 0.5 * (1.0 + math.cos(math.pi * progress))\n            min_lr = base_lr * 0.01\n            return min_lr + (base_lr - min_lr) * cosine_val\n \n    return callbacks.LearningRateScheduler(schedule, verbose=0)\n ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-12T15:57:59.555328Z","iopub.execute_input":"2026-05-12T15:57:59.555637Z","iopub.status.idle":"2026-05-12T15:57:59.568671Z","shell.execute_reply.started":"2026-05-12T15:57:59.555614Z","shell.execute_reply":"2026-05-12T15:57:59.567812Z"}},"outputs":[],"execution_count":null},{"id":"d9fcb6da","cell_type":"markdown","source":"#### Updated constants","metadata":{}},{"id":"695df322","cell_type":"code","source":"\nBASE_LR = 1e-4    # unchanged from your current value\n \n# 40 epochs: with 200k samples and GLOBAL_BATCH_SIZE=128, one epoch =\n# ~1562 steps. 40 epochs ≈ 62k gradient updates — enough for PASH\n# prototypes to converge. EarlyStopping(patience=8) means if the model\n# genuinely plateaus it will stop before hitting 40.\n# Rough Kaggle T4x2 estimate: ~10-12 min/epoch → 40 epochs ≈ 7-8 h per fold.\n# With 5 folds that exceeds one session, so use ModelCheckpoint to resume.\nEPOCHS = 40\n \nWARMUP_EPOCHS = 3   # 3/40 ≈ 7% warmup — standard for transformer components\n \nMETRICS = [\n    'accuracy',\n    tf.keras.metrics.AUC(name='roc_auc', curve='ROC')\n]\n \n \ndef evaluate_model(model, dataset):\n    \"\"\"Unchanged from your original — kept here for completeness.\"\"\"\n    probs = model.predict(dataset, verbose=0).ravel()\n    labels_arr = np.concatenate([y.numpy() for _, y in dataset], axis=0)\n    preds = (probs > 0.5).astype(int)\n    return {\n        'accuracy': accuracy_score(labels_arr, preds),\n        'roc_auc':  roc_auc_score(labels_arr, probs)\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-12T15:57:59.570741Z","iopub.execute_input":"2026-05-12T15:57:59.571104Z","iopub.status.idle":"2026-05-12T15:57:59.587875Z","shell.execute_reply.started":"2026-05-12T15:57:59.571072Z","shell.execute_reply":"2026-05-12T15:57:59.587229Z"}},"outputs":[],"execution_count":null},{"id":"fdb43d2f","cell_type":"markdown","source":"**Stratified K-FOLD Validation**","metadata":{"papermill":{"duration":0.432234,"end_time":"2026-05-06T00:35:37.041407+00:00","exception":false,"start_time":"2026-05-06T00:35:36.609173+00:00","status":"completed"},"tags":[]}},{"id":"53ea75eb","cell_type":"markdown","source":"Changes to take advantage of the 2 GPUs","metadata":{"papermill":{"duration":0.432211,"end_time":"2026-05-06T00:35:37.908109+00:00","exception":false,"start_time":"2026-05-06T00:35:37.475898+00:00","status":"completed"},"tags":[]}},{"id":"3e778fff","cell_type":"code","source":"import tensorflow as tf\n\n# For using the 2 GPUs available\nstrategy = tf.distribute.MirroredStrategy()\nprint(f'Detected Devices: {strategy.num_replicas_in_sync}') \n\n# Scale Batch Size\n# Each GPU will process 64 images per batch giving a total of 128\nBATCH_SIZE_PER_REPLICA = 64 \nGLOBAL_BATCH_SIZE = BATCH_SIZE_PER_REPLICA * strategy.num_replicas_in_sync","metadata":{"execution":{"iopub.status.busy":"2026-05-12T15:57:59.588778Z","iopub.execute_input":"2026-05-12T15:57:59.588956Z","iopub.status.idle":"2026-05-12T15:57:59.603776Z","shell.execute_reply.started":"2026-05-12T15:57:59.588938Z","shell.execute_reply":"2026-05-12T15:57:59.602827Z"},"papermill":{"duration":0.457205,"end_time":"2026-05-06T00:35:38.795902+00:00","exception":false,"start_time":"2026-05-06T00:35:38.338697+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"1af6537b","cell_type":"markdown","source":"## Clahe Preprocessing","metadata":{}},{"id":"8d85e082","cell_type":"code","source":"_CLAHE = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-12T15:57:59.604732Z","iopub.execute_input":"2026-05-12T15:57:59.605303Z","iopub.status.idle":"2026-05-12T15:57:59.609979Z","shell.execute_reply.started":"2026-05-12T15:57:59.605282Z","shell.execute_reply":"2026-05-12T15:57:59.609199Z"}},"outputs":[],"execution_count":null},{"id":"e4737244","cell_type":"code","source":"def _read_and_clahe(path_str: str) -> np.ndarray:\n    \"\"\"\n    Shared logic for both train and test loaders.\n    Reads a .tif file, converts BGR→RGB, applies CLAHE per channel,\n    then min-max normalizes to [0, 1] float32.\n    \"\"\"\n    img = cv2.imread(path_str)          # (96, 96, 3) uint8, BGR\n    if img is None:\n        return np.zeros((*IMG_SIZE, 3), dtype=np.float32)\n \n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n \n    # CLAHE.apply() requires a single-channel uint8 array — run per channel\n    channels = [_CLAHE.apply(img[:, :, c]) for c in range(img.shape[2])]\n    enhanced = np.stack(channels, axis=-1)          # (96, 96, 3) uint8\n \n    return enhanced.astype(np.float32) / 255.0      # [0, 1] float32\n \n \ndef load_train_clahe(path: tf.Tensor, label: tf.Tensor):\n    \"\"\"\n    Drop-in replacement for your load_train function.\n    tf.py_function is required because cv2 cannot run in TF graph mode.\n    set_shape() tells the model graph the static input dimensions —\n    without it, downstream layers see None dims and may error.\n    \"\"\"\n    def _fn(p):\n        return _read_and_clahe(p.numpy().decode(\"utf-8\"))\n \n    img = tf.py_function(_fn, [path], tf.float32)\n    img.set_shape([*IMG_SIZE, 3])\n    return img, label\n \n \ndef decode_image_clahe(path: tf.Tensor) -> tf.Tensor:\n    \"\"\"\n    Drop-in replacement for your decode_image function (test set).\n    Same CLAHE pipeline — train/test preprocessing must be identical.\n    \"\"\"\n    def _fn(p):\n        return _read_and_clahe(p.numpy().decode(\"utf-8\"))\n \n    img = tf.py_function(_fn, [path], tf.float32)\n    img.set_shape([*IMG_SIZE, 3])\n    return img\n \n \ndef make_dataset(files, labels=None, training=False, batch_size=None):\n    \"\"\"\n    Updated make_dataset. Signature is identical to yours so the\n    training loop below works without any other changes.\n \n    batch_size should always be GLOBAL_BATCH_SIZE (not per-replica).\n    MirroredStrategy splits the global batch across GPUs automatically.\n \n    Shuffle buffer = 4096:\n      A full-dataset buffer (200k images) would cost ~4.4 GB RAM at\n      float32 96×96×3. 4096 gives adequate randomness at ~90 MB.\n    \"\"\"\n    if batch_size is None:\n        raise ValueError(\n            \"Always pass batch_size=GLOBAL_BATCH_SIZE explicitly. \"\n            \"MirroredStrategy will shard it across GPUs automatically.\"\n        )\n \n    if labels is None:\n        ds = tf.data.Dataset.from_tensor_slices(files)\n        ds = ds.map(decode_image_clahe, num_parallel_calls=tf.data.AUTOTUNE)\n    else:\n        ds = tf.data.Dataset.from_tensor_slices((files, labels))\n        ds = ds.map(load_train_clahe, num_parallel_calls=tf.data.AUTOTUNE)\n \n    if training:\n        ds = ds.shuffle(4096, seed=SEED, reshuffle_each_iteration=True)\n \n    # batch() with GLOBAL_BATCH_SIZE — MirroredStrategy shards internally\n    return ds.batch(batch_size).prefetch(tf.data.AUTOTUNE)\n ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-12T15:57:59.610878Z","iopub.execute_input":"2026-05-12T15:57:59.611168Z","iopub.status.idle":"2026-05-12T15:57:59.623787Z","shell.execute_reply.started":"2026-05-12T15:57:59.611118Z","shell.execute_reply":"2026-05-12T15:57:59.623017Z"}},"outputs":[],"execution_count":null},{"id":"cf6ec1b8","cell_type":"markdown","source":"## HYBRID MODEL ResNet-ViT + BDCG + PASH","metadata":{"papermill":{"duration":0.485118,"end_time":"2026-05-06T00:35:40.582726+00:00","exception":false,"start_time":"2026-05-06T00:35:40.097608+00:00","status":"completed"},"tags":[]}},{"id":"4f59d138","cell_type":"code","source":"# BASIC COMPONENTS (RESNET & TRANSFORMER)\n\ndef residual_block(x, filters, kernel_size=3):\n    # RESIDUAL BLOCK\n    # Extract local and resolutive features while maintaining gradient flow, avoid vanishing\n\n    shortcut = x\n    x = layers.Conv2D(filters, kernel_size, padding='same')(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.Activation('relu')(x)\n    x = layers.Conv2D(filters, kernel_size, padding='same')(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.Add()([x, shortcut])\n    x = layers.Activation('relu')(x)\n    return x\n\ndef mlp_block(x, hidden_dim, dropout_rate=0.1):\n    # TRANSFORMER MLP BLOCK\n    # MLP like in a 4 Layer Transformer Encoder\n    # This does an expansion-and-compression bottleneck\n\n    # We save the original dimension, in this case 64\n    original_dim = x.shape[-1] \n    \n    # First projects features to higher dimensional space (hidden dim, 128) to learn complex interactions using Gaussian Error Linear Unit\n    x = layers.Dense(hidden_dim, activation=tf.nn.gelu)(x)\n    x = layers.Dropout(dropout_rate)(x)\n    \n    # Then projects bacjk to original dimension to maintain compatibility with skip connection ( Residual )\n    x = layers.Dense(original_dim)(x)\n    x = layers.Dropout(dropout_rate)(x)\n    \n    return x\n\n# ADVANCED LAYERS\n\nclass BDCGFusionLayer(layers.Layer):\n    \"\"\"\n    Bi-Directional Cross-Guidance (BDCG)\n    This allows the mutual refinement between local (CNN) and global (ViT) features.\n    \"\"\"\n    def __init__(self, embed_dim=64, num_heads=4, **kwargs):\n        super(BDCGFusionLayer, self).__init__(**kwargs)\n        self.embed_dim = embed_dim\n        \n        # Atención de Global a Local: ViT guía a la CNN\n        self.mha_g2l = layers.MultiHeadAttention(num_heads=num_heads, key_dim=embed_dim)\n        # Atención de Local a Global: CNN guía al ViT\n        self.mha_l2g = layers.MultiHeadAttention(num_heads=num_heads, key_dim=embed_dim)\n        \n        self.norm_l = layers.LayerNormalization(epsilon=1e-6)\n        self.norm_g = layers.LayerNormalization(epsilon=1e-6)\n\n    def call(self, local_seq, global_seq):\n        # El contexto anatómico (ViT) valida las texturas (CNN)\n        refined_local = self.mha_g2l(query=local_seq, value=global_seq, key=global_seq)\n        out_local = self.norm_l(local_seq + refined_local)\n        \n        # Las estructuras a nivel de lesión (CNN) restringen el razonamiento global (ViT)\n        refined_global = self.mha_l2g(query=global_seq, value=local_seq, key=local_seq)\n        out_global = self.norm_g(global_seq + refined_global)\n        \n        return out_local, out_global\n\nclass PASHClassifier(layers.Layer):\n    \"\"\"\n    Prototype-Anchored Similarity Head (PASH)\n    Clasificación basada en distancia semántica usando una distribución t-Student.\n    \"\"\"\n    def __init__(self, embed_dim=128, **kwargs):\n        super(PASHClassifier, self).__init__(**kwargs)\n        self.embed_dim = embed_dim\n\n    def build(self, input_shape):\n        # Prototipos aprendibles: [0] = Benigno, [1] = Maligno\n        self.prototypes = self.add_weight(\n            shape=(2, self.embed_dim),\n            initializer=\"glorot_uniform\",\n            trainable=True,\n            name=\"class_prototypes\"\n        )\n        # Grados de libertad (alpha) para la distribución t-Student\n        self.alpha = self.add_weight(\n            shape=(1,),\n            initializer=tf.keras.initializers.Constant(3.0),\n            trainable=True,\n            name=\"t_student_alpha\"\n        )\n\n    def call(self, z):\n        # z shape: (Batch, embed_dim)\n        # 1. Calcular distancia Euclidiana al cuadrado a cada prototipo\n        # (Batch, 1, embed_dim) - (1, 2, embed_dim) -> (Batch, 2, embed_dim)\n        z_expanded = tf.expand_dims(z, axis=1)\n        p_expanded = tf.expand_dims(self.prototypes, axis=0)\n        \n        # Distancias: shape (Batch, 2)\n        distances = tf.reduce_sum(tf.square(z_expanded - p_expanded), axis=-1)\n        \n        # 2. Convertir distancia a probabilidad con kernel t-Student\n        # P(y=k|X) = (1 + d / alpha) ^ -((alpha + 1) / 2)\n        exponent = -(self.alpha + 1.0) / 2.0\n        similarity = tf.pow(1.0 + (distances / self.alpha), exponent)\n        \n        # Normalizar para obtener probabilidades que sumen 1 (Softmax sobre similitudes)\n        prob_dist = similarity / tf.reduce_sum(similarity, axis=-1, keepdims=True)\n        \n        # Retornar la probabilidad de la clase maligna (índice 1) para compatibilidad con BinaryCrossentropy\n        return prob_dist[:, 1:]\n\n\n# HYBRID PRINCIPAL MODEL\n\n\ndef build_hybrid_model(input_shape=(96, 96, 3), embed_dim=64, patch_size=12, lr=1e-4):\n    inputs = layers.Input(shape=input_shape)\n    \n    # RAMA LOCAL (CNN - ResNet) \n    # Extrae texturas de grano fino y bordes de lesiones\n    x = layers.Conv2D(embed_dim, (7, 7), strides=2, padding='same')(inputs)\n    x = layers.BatchNormalization()(x)\n    x = layers.Activation('relu')(x)\n    x = layers.MaxPooling2D((3, 3), strides=2, padding='same')(x)\n    \n    x = residual_block(x, embed_dim)\n    cnn_features = residual_block(x, embed_dim) # Output: (24, 24, 64)\n    \n    # Aplanar mapas 2D a secuencia 1D: (Batch, 576, 64)\n    _, h, w, c = cnn_features.shape\n    local_seq = layers.Reshape((h * w, c))(cnn_features)\n    \n    # RAMA GLOBAL (Vision Transformer Ligero) \n    # Extrae dependencias a largo plazo y contexto anatómico\n    # Usamos Conv2D como un truco rápido y eficiente para extraer parches\n    patches = layers.Conv2D(embed_dim, kernel_size=patch_size, strides=patch_size)(inputs)\n    _, ph, pw, pc = patches.shape\n    global_seq = layers.Reshape((ph * pw, pc))(patches) # Output: (64, 64)\n    \n    # Positional Encoding simple para el ViT\n    positions = tf.range(start=0, limit=ph * pw, delta=1)\n    pos_emb = layers.Embedding(input_dim=ph * pw, output_dim=embed_dim)(positions)\n    global_seq = global_seq + pos_emb\n    \n    # 1 Bloque de Transformer Encoder\n    attn_out = layers.MultiHeadAttention(num_heads=4, key_dim=embed_dim)(global_seq, global_seq)\n    global_seq = layers.LayerNormalization(epsilon=1e-6)(global_seq + attn_out)\n    mlp_out = mlp_block(global_seq, hidden_dim=embed_dim * 2)\n    global_seq = layers.LayerNormalization(epsilon=1e-6)(global_seq + mlp_out)\n\n    # FUSIÓN BDCG \n    # Alinea las características locales y globales interactuando entre sí\n    bdcg_layer = BDCGFusionLayer(embed_dim=embed_dim, num_heads=4)\n    refined_local, refined_global = bdcg_layer(local_seq, global_seq)\n    \n    # Global Average Pooling para ambas secuencias\n    pool_local = layers.GlobalAveragePooling1D()(refined_local)\n    pool_global = layers.GlobalAveragePooling1D()(refined_global)\n    \n    # Representación unificada (128 dimensiones)\n    fused_representation = layers.Concatenate()([pool_local, pool_global])\n    fused_representation = layers.Dropout(0.3)(fused_representation)\n    \n    # CLASIFICADOR PASH\n    # Toma decisiones basadas en similitud a prototipos clínicos aprendidos\n    outputs = PASHClassifier(embed_dim=embed_dim * 2)(fused_representation)\n    \n    model = models.Model(inputs, outputs, name=\"Hybrid_CNN_ViT_BDCG_PASH\")\n    \n    # Compilación\n    model.compile(\n        optimizer=optimizers.Adam(learning_rate=lr, beta_1=0.9, beta_2=0.999),\n        loss='binary_crossentropy',\n        metrics=['accuracy', tf.keras.metrics.AUC(name='roc_auc')]\n    )\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2026-05-12T15:57:59.624769Z","iopub.execute_input":"2026-05-12T15:57:59.625608Z","iopub.status.idle":"2026-05-12T15:57:59.646467Z","shell.execute_reply.started":"2026-05-12T15:57:59.625574Z","shell.execute_reply":"2026-05-12T15:57:59.645870Z"},"papermill":{"duration":0.536404,"end_time":"2026-05-06T00:35:41.598147+00:00","exception":false,"start_time":"2026-05-06T00:35:41.061743+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"8459c500","cell_type":"code","source":"# CELL 4 — Updated K-Fold training loop\n# The only changes vs your original:\n#   1. make_dataset now calls CLAHE loaders\n#   2. callbacks include cosine schedule instead of ReduceLROnPlateau\n#   3. ModelCheckpoint added as crash recovery\n#   4. epochs=EPOCHS (40) instead of 8\n# Everything else — strategy.scope(), skf.split(), files_arr, labels_arr,\n# build_hybrid_model — is exactly as your notebook defines it.\n# =============================================================================\n \n#K = 1 # Number of folds\n#skf = StratifiedKFold(n_splits=K, shuffle=True, random_state=SEED)\n\n#all_fold_results = []\n \n#for fold, (train_idx, val_idx) in enumerate(skf.split(files_arr, labels_arr)):\n#    print(f\"\\n{'='*60}\")\n#    print(f\"  FOLD {fold + 1} / {K}  |  CLAHE + Cosine LR  |  {EPOCHS} epochs\")\n#    print(f\"{'='*60}\")\n \n#    x_train_fold = files_arr[train_idx]\n#    x_val_fold   = files_arr[val_idx]\n#     y_train_fold = labels_arr[train_idx]\n#    y_val_fold   = labels_arr[val_idx]\n \n    # GLOBAL_BATCH_SIZE (=128) is passed to ds.batch().\n    # MirroredStrategy will split the 128 into 64 per GPU automatically.\n#    train_ds = make_dataset(\n#        x_train_fold, y_train_fold,\n#        training=True,\n#        batch_size=GLOBAL_BATCH_SIZE   # ← your existing variable, unchanged\n#    )\n#    val_ds = make_dataset(\n#        x_val_fold, y_val_fold,\n#        training=False,\n#        batch_size=GLOBAL_BATCH_SIZE   # ← same for val; no shuffling applied\n#    )\n \n    # Callbacks assembled per fold so the checkpoint filepath is fold-specific\n#    fold_callbacks = [\n#        callbacks.EarlyStopping(\n#            monitor='val_roc_auc',\n#            patience=8,         # up from 5 — gives cosine schedule room to work\n#            mode='max',\n#            restore_best_weights=True,\n#            verbose=1\n#        ),\n#        make_cosine_schedule(\n#            total_epochs=EPOCHS,\n#            warmup_epochs=WARMUP_EPOCHS,\n#            base_lr=BASE_LR\n#        ),\n#        callbacks.ModelCheckpoint(\n#            filepath=f\"/kaggle/working/best_fold{fold + 1}.keras\",\n#            monitor='val_roc_auc',\n#            save_best_only=True,\n#            mode='max',\n#            verbose=0           # silent — reduces log noise over 40 epochs\n#        ),\n#    ]\n# \n#    # Model built inside strategy.scope() — same as your original\n#    with strategy.scope():\n#        model_hybrid = build_hybrid_model(input_shape=(96, 96, 3), lr=BASE_LR)\n \n#    model_hybrid.fit(\n#        train_ds,\n#        validation_data=val_ds,\n#        epochs=EPOCHS,\n#        callbacks=fold_callbacks,\n#        verbose=1\n#    )\n \n#    metrics = evaluate_model(model_hybrid, val_ds)\n#    all_fold_results.append(metrics)\n#    print(f\"\\nFold {fold + 1}  →  Accuracy: {metrics['accuracy']:.4f}  |  AUC: {metrics['roc_auc']:.4f}\")\n \n#    model_hybrid.save(f\"/kaggle/working/hybrid_model_fold{fold + 1}.keras\")\n#    print(f\"Saved: hybrid_model_fold{fold + 1}.keras\")\n \n# ── Summary ──────────────────────────────────────────────────────────────────\n#mean_accuracy = np.mean([m['accuracy'] for m in all_fold_results])\n#mean_auc      = np.mean([m['roc_auc']  for m in all_fold_results])\n#print(f\"\\n{'='*60}\")\n#print(f\"  K-Fold mean  →  Accuracy: {mean_accuracy:.4f}  |  AUC: {mean_auc:.4f}\")\n#print(f\"{'='*60}\")\n ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-12T15:57:59.647344Z","iopub.execute_input":"2026-05-12T15:57:59.647647Z","iopub.status.idle":"2026-05-12T15:57:59.662548Z","shell.execute_reply.started":"2026-05-12T15:57:59.647619Z","shell.execute_reply":"2026-05-12T15:57:59.661799Z"}},"outputs":[],"execution_count":null},{"id":"833be5a7-e533-4014-8a80-58e054aacf64","cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nprint(f\"\\n{'='*60}\")\nprint(f\"  SINGLE RUN  |  CLAHE + Cosine LR  |  {EPOCHS} epochs\")\n#print(f\"{='*60}\")\n\n# 1. Do a standard 80/20 split instead of K-Fold\nx_train_fold, x_val_fold, y_train_fold, y_val_fold = train_test_split(\n    files_arr, labels_arr, test_size=0.2, stratify=labels_arr, random_state=SEED\n)\n\n# 2. Create datasets\ntrain_ds = make_dataset(\n    x_train_fold, y_train_fold,\n    training=True,\n    batch_size=GLOBAL_BATCH_SIZE   \n)\nval_ds = make_dataset(\n    x_val_fold, y_val_fold,\n    training=False,\n    batch_size=GLOBAL_BATCH_SIZE   \n)\n\n# 3. Callbacks\nfold_callbacks = [\n    callbacks.EarlyStopping(\n        monitor='val_roc_auc',\n        patience=8,         \n        mode='max',\n        restore_best_weights=True,\n        verbose=1\n    ),\n    make_cosine_schedule(\n        total_epochs=EPOCHS,\n        warmup_epochs=WARMUP_EPOCHS,\n        base_lr=BASE_LR\n    ),\n    callbacks.ModelCheckpoint(\n        filepath=\"/kaggle/working/best_single_run.keras\",\n        monitor='val_roc_auc',\n        save_best_only=True,\n        mode='max',\n        verbose=0           \n    ),\n]\n\n# 4. Build and Train\nwith strategy.scope():\n    model_hybrid = build_hybrid_model(input_shape=(96, 96, 3), lr=BASE_LR)\n\nmodel_hybrid.fit(\n    train_ds,\n    validation_data=val_ds,\n    epochs=EPOCHS,\n    callbacks=fold_callbacks,\n    verbose=1\n)\n\nmetrics = evaluate_model(model_hybrid, val_ds)\nprint(f\"\\nFinal ->  Accuracy: {metrics['accuracy']:.4f}  |  AUC: {metrics['roc_auc']:.4f}\")\n\nmodel_hybrid.save(\"/kaggle/working/hybrid_model_4CONF_CLAHE_Tunned.keras\")\nprint(\"Saved: hybrid_model_4CONF_CLAHE_Tunned.keras\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}