{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":11848,"databundleVersionId":862157}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport cv2\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras import layers, models","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:49:25.791852Z","iopub.execute_input":"2026-04-12T09:49:25.792305Z","iopub.status.idle":"2026-04-12T09:49:25.796966Z","shell.execute_reply.started":"2026-04-12T09:49:25.792247Z","shell.execute_reply":"2026-04-12T09:49:25.795976Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Loading Labels","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\nlabels = pd.read_csv(\n    \"/kaggle/input/competitions/histopathologic-cancer-detection/train_labels.csv\"\n)\n\nprint(labels.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:49:25.798386Z","iopub.execute_input":"2026-04-12T09:49:25.799053Z","iopub.status.idle":"2026-04-12T09:49:26.034975Z","shell.execute_reply.started":"2026-04-12T09:49:25.799028Z","shell.execute_reply":"2026-04-12T09:49:26.034044Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_path = \"/kaggle/input/competitions/histopathologic-cancer-detection/train/\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:49:26.035910Z","iopub.execute_input":"2026-04-12T09:49:26.036147Z","iopub.status.idle":"2026-04-12T09:49:26.040561Z","shell.execute_reply.started":"2026-04-12T09:49:26.036107Z","shell.execute_reply":"2026-04-12T09:49:26.039679Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nprint(os.listdir(\"/kaggle/input/competitions/histopathologic-cancer-detection\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:49:26.042586Z","iopub.execute_input":"2026-04-12T09:49:26.042918Z","iopub.status.idle":"2026-04-12T09:49:26.053802Z","shell.execute_reply.started":"2026-04-12T09:49:26.042883Z","shell.execute_reply":"2026-04-12T09:49:26.052836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\nsample_size = 5000  # safe start\nlabels_sample = labels.sample(sample_size, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:49:26.054940Z","iopub.execute_input":"2026-04-12T09:49:26.055317Z","iopub.status.idle":"2026-04-12T09:49:26.079839Z","shell.execute_reply.started":"2026-04-12T09:49:26.055276Z","shell.execute_reply":"2026-04-12T09:49:26.079156Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\n\ntrain_path = \"/kaggle/input/competitions/histopathologic-cancer-detection/train/\"\n\ndef load_images(df):\n    images = []\n    y = []\n    \n    for _, row in df.iterrows():\n        img_path = os.path.join(train_path, row[\"id\"] + \".tif\")\n        \n        img = cv2.imread(img_path)\n        img = cv2.resize(img, (96, 96))\n        img = img / 255.0\n        \n        images.append(img)\n        y.append(row[\"label\"])\n    \n    return np.array(images), np.array(y)\n\nX, y = load_images(labels_sample)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:49:26.080991Z","iopub.execute_input":"2026-04-12T09:49:26.081738Z","iopub.status.idle":"2026-04-12T09:49:33.301771Z","shell.execute_reply.started":"2026-04-12T09:49:26.081706Z","shell.execute_reply":"2026-04-12T09:49:33.301048Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Safety check","metadata":{}},{"cell_type":"code","source":"print(X.shape, y.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:49:33.302728Z","iopub.execute_input":"2026-04-12T09:49:33.303067Z","iopub.status.idle":"2026-04-12T09:49:33.307349Z","shell.execute_reply.started":"2026-04-12T09:49:33.303022Z","shell.execute_reply":"2026-04-12T09:49:33.306331Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Building the cnn modoel","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras import layers, models\n\nmodel = models.Sequential([\n    layers.Conv2D(32, (3,3), activation='relu', input_shape=(96,96,3)),\n    layers.MaxPooling2D(2,2),\n\n    layers.Conv2D(64, (3,3), activation='relu'),\n    layers.MaxPooling2D(2,2),\n\n    layers.Flatten(),\n    layers.Dense(64, activation='relu'),\n    layers.Dense(1, activation='sigmoid')\n])\n\nmodel.compile(\n    optimizer='adam',\n    loss='binary_crossentropy',\n    metrics=['accuracy']\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:49:33.308410Z","iopub.execute_input":"2026-04-12T09:49:33.308640Z","iopub.status.idle":"2026-04-12T09:49:33.358420Z","shell.execute_reply.started":"2026-04-12T09:49:33.308621Z","shell.execute_reply":"2026-04-12T09:49:33.357661Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Training the model","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_val, y_train, y_val = train_test_split(\n    X, y, test_size=0.2, random_state=42\n)\n\nhistory = model.fit(\n    X_train, y_train,\n    epochs=5,\n    validation_data=(X_val, y_val)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:49:33.360561Z","iopub.execute_input":"2026-04-12T09:49:33.360909Z","iopub.status.idle":"2026-04-12T09:49:44.940408Z","shell.execute_reply.started":"2026-04-12T09:49:33.360885Z","shell.execute_reply":"2026-04-12T09:49:44.939444Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Evaluating the model","metadata":{}},{"cell_type":"code","source":"model.evaluate(X_val, y_val)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:49:44.941419Z","iopub.execute_input":"2026-04-12T09:49:44.941672Z","iopub.status.idle":"2026-04-12T09:49:45.484870Z","shell.execute_reply.started":"2026-04-12T09:49:44.941648Z","shell.execute_reply":"2026-04-12T09:49:45.484145Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Importing the pretrained model","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.applications import MobileNetV2\nfrom tensorflow.keras import layers, models","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:49:45.485760Z","iopub.execute_input":"2026-04-12T09:49:45.486178Z","iopub.status.idle":"2026-04-12T09:49:45.490285Z","shell.execute_reply.started":"2026-04-12T09:49:45.486151Z","shell.execute_reply":"2026-04-12T09:49:45.489275Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"base_model = MobileNetV2(\n    weights='imagenet',\n    include_top=False,\n    input_shape=(96,96,3)\n)\n\nbase_model.trainable = False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:49:45.491160Z","iopub.execute_input":"2026-04-12T09:49:45.491445Z","iopub.status.idle":"2026-04-12T09:49:46.202873Z","shell.execute_reply.started":"2026-04-12T09:49:45.491410Z","shell.execute_reply":"2026-04-12T09:49:46.202038Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = models.Sequential([\n    base_model,\n    layers.GlobalAveragePooling2D(),\n    layers.Dense(64, activation='relu'),\n    layers.Dropout(0.3),\n    layers.Dense(1, activation='sigmoid')\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:49:46.203936Z","iopub.execute_input":"2026-04-12T09:49:46.204224Z","iopub.status.idle":"2026-04-12T09:49:46.230286Z","shell.execute_reply.started":"2026-04-12T09:49:46.204200Z","shell.execute_reply":"2026-04-12T09:49:46.229724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.compile(\n    optimizer='adam',\n    loss='binary_crossentropy',\n    metrics=['accuracy']\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:49:46.231111Z","iopub.execute_input":"2026-04-12T09:49:46.231413Z","iopub.status.idle":"2026-04-12T09:49:46.239306Z","shell.execute_reply.started":"2026-04-12T09:49:46.231390Z","shell.execute_reply":"2026-04-12T09:49:46.238573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = model.fit(\n    X_train, y_train,\n    epochs=5,\n    validation_data=(X_val, y_val)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:49:46.240213Z","iopub.execute_input":"2026-04-12T09:49:46.240407Z","iopub.status.idle":"2026-04-12T09:50:10.181157Z","shell.execute_reply.started":"2026-04-12T09:49:46.240387Z","shell.execute_reply":"2026-04-12T09:50:10.180196Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Testing accuracy","metadata":{}},{"cell_type":"code","source":"val_loss, val_acc = model.evaluate(X_val, y_val)\nprint(\"Validation Accuracy:\", val_acc)\nprint(\"Validation Loss:\", val_loss)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:50:10.182267Z","iopub.execute_input":"2026-04-12T09:50:10.182653Z","iopub.status.idle":"2026-04-12T09:50:10.887153Z","shell.execute_reply.started":"2026-04-12T09:50:10.182624Z","shell.execute_reply":"2026-04-12T09:50:10.886350Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Confusion Matric","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# predictions\ny_pred = (model.predict(X_val) > 0.5).astype(\"int32\")\n\n# confusion matrix\ncm = confusion_matrix(y_val, y_pred)\n\nplt.figure(figsize=(5,4))\nsns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\")\nplt.xlabel(\"Predicted\")\nplt.ylabel(\"Actual\")\nplt.title(\"Confusion Matrix - Cancer Detection\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:50:10.888181Z","iopub.execute_input":"2026-04-12T09:50:10.888400Z","iopub.status.idle":"2026-04-12T09:50:19.569850Z","shell.execute_reply.started":"2026-04-12T09:50:10.888378Z","shell.execute_reply":"2026-04-12T09:50:19.569049Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"PRECISION, RECALL, F1","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import classification_report\n\nprint(classification_report(y_val, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:50:19.571040Z","iopub.execute_input":"2026-04-12T09:50:19.571379Z","iopub.status.idle":"2026-04-12T09:50:19.587650Z","shell.execute_reply.started":"2026-04-12T09:50:19.571342Z","shell.execute_reply":"2026-04-12T09:50:19.586773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc\nimport matplotlib.pyplot as plt\n\ny_prob = model.predict(X_val).ravel()\n\nfpr, tpr, _ = roc_curve(y_val, y_prob)\nroc_auc = auc(fpr, tpr)\n\nplt.figure()\nplt.plot(fpr, tpr, label=f\"AUC = {roc_auc:.4f}\")\nplt.plot([0,1], [0,1], linestyle=\"--\")\nplt.xlabel(\"False Positive Rate\")\nplt.ylabel(\"True Positive Rate\")\nplt.title(\"ROC Curve\")\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:50:19.588911Z","iopub.execute_input":"2026-04-12T09:50:19.589310Z","iopub.status.idle":"2026-04-12T09:50:20.476434Z","shell.execute_reply.started":"2026-04-12T09:50:19.589254Z","shell.execute_reply":"2026-04-12T09:50:20.475739Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\n\nwrong_idx = np.where(y_pred.flatten() != y_val)[0]\n\nprint(\"Misclassified samples:\", len(wrong_idx))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:50:20.477673Z","iopub.execute_input":"2026-04-12T09:50:20.478011Z","iopub.status.idle":"2026-04-12T09:50:20.486381Z","shell.execute_reply.started":"2026-04-12T09:50:20.477986Z","shell.execute_reply":"2026-04-12T09:50:20.485374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in wrong_idx[:5]:\n    plt.imshow(X_val[i])\n    plt.title(f\"True: {y_val[i]} | Pred: {y_pred[i][0]}\")\n    plt.axis(\"off\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:50:20.487422Z","iopub.execute_input":"2026-04-12T09:50:20.487784Z","iopub.status.idle":"2026-04-12T09:50:20.793258Z","shell.execute_reply.started":"2026-04-12T09:50:20.487760Z","shell.execute_reply":"2026-04-12T09:50:20.792260Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport cv2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:50:20.794455Z","iopub.execute_input":"2026-04-12T09:50:20.794867Z","iopub.status.idle":"2026-04-12T09:50:20.799732Z","shell.execute_reply.started":"2026-04-12T09:50:20.794797Z","shell.execute_reply":"2026-04-12T09:50:20.799116Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.applications.mobilenet_v2 import preprocess_input\n\nimg = preprocess_input(X_val[0:1].copy())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:50:20.800674Z","iopub.execute_input":"2026-04-12T09:50:20.801144Z","iopub.status.idle":"2026-04-12T09:50:20.810959Z","shell.execute_reply.started":"2026-04-12T09:50:20.801118Z","shell.execute_reply":"2026-04-12T09:50:20.810263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\n\ndef get_gradcam(img_array, model, layer_name):\n\n    base_model = model.layers[0]  # MobileNetV2 inside\n\n    grad_model = tf.keras.models.Model(\n        inputs=model.input,\n        outputs=[\n            base_model.get_layer(layer_name).output,\n            model.output\n        ]\n    )\n\n    with tf.GradientTape() as tape:\n        conv_outputs, predictions = grad_model(img_array)\n        loss = predictions[:, 0]\n\n    grads = tape.gradient(loss, conv_outputs)\n\n    pooled_grads = tf.reduce_mean(grads, axis=(0, 1, 2))\n\n    conv_outputs = conv_outputs[0]\n\n    heatmap = conv_outputs @ pooled_grads[..., tf.newaxis]\n    heatmap = tf.squeeze(heatmap)\n\n    heatmap = tf.nn.relu(heatmap)\n    heatmap = heatmap / (tf.reduce_max(heatmap) + 1e-8)\n\n    return heatmap.numpy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:50:20.814656Z","iopub.execute_input":"2026-04-12T09:50:20.814988Z","iopub.status.idle":"2026-04-12T09:50:20.823587Z","shell.execute_reply.started":"2026-04-12T09:50:20.814956Z","shell.execute_reply":"2026-04-12T09:50:20.822711Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install tf-keras-vis","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:50:20.824531Z","iopub.execute_input":"2026-04-12T09:50:20.824842Z","iopub.status.idle":"2026-04-12T09:50:24.342044Z","shell.execute_reply.started":"2026-04-12T09:50:20.824791Z","shell.execute_reply":"2026-04-12T09:50:24.341164Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nfrom tf_keras_vis.gradcam import Gradcam\nfrom tf_keras_vis.utils.model_modifiers import ReplaceToLinear\nfrom tf_keras_vis.utils.scores import CategoricalScore","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:50:24.343555Z","iopub.execute_input":"2026-04-12T09:50:24.343929Z","iopub.status.idle":"2026-04-12T09:50:24.350684Z","shell.execute_reply.started":"2026-04-12T09:50:24.343887Z","shell.execute_reply":"2026-04-12T09:50:24.349899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"replace2linear = ReplaceToLinear()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:50:24.351776Z","iopub.execute_input":"2026-04-12T09:50:24.352180Z","iopub.status.idle":"2026-04-12T09:50:24.362953Z","shell.execute_reply.started":"2026-04-12T09:50:24.352141Z","shell.execute_reply":"2026-04-12T09:50:24.362006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\n\nbase_model = model.layers[0]\n\ninputs = tf.keras.Input(shape=(96,96,3))\n\nx = base_model(inputs, training=False)\nx = model.layers[1](x)\nx = model.layers[2](x)\nx = model.layers[3](x)\noutputs = model.layers[4](x)\n\nclean_model = tf.keras.Model(inputs, outputs)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:50:24.363983Z","iopub.execute_input":"2026-04-12T09:50:24.364318Z","iopub.status.idle":"2026-04-12T09:50:24.386591Z","shell.execute_reply.started":"2026-04-12T09:50:24.364294Z","shell.execute_reply":"2026-04-12T09:50:24.385696Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tf_keras_vis.gradcam import Gradcam\nfrom tf_keras_vis.utils.model_modifiers import ReplaceToLinear\nfrom tf_keras_vis.utils.scores import BinaryScore","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:50:24.388917Z","iopub.execute_input":"2026-04-12T09:50:24.389123Z","iopub.status.idle":"2026-04-12T09:50:24.393318Z","shell.execute_reply.started":"2026-04-12T09:50:24.389103Z","shell.execute_reply":"2026-04-12T09:50:24.392274Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"replace2linear = ReplaceToLinear()\n\ngradcam = Gradcam(clean_model, model_modifier=replace2linear, clone=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:50:24.394412Z","iopub.execute_input":"2026-04-12T09:50:24.394722Z","iopub.status.idle":"2026-04-12T09:50:25.747310Z","shell.execute_reply.started":"2026-04-12T09:50:24.394688Z","shell.execute_reply":"2026-04-12T09:50:25.746370Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"score = BinaryScore(0)  # class 0 or 1 depending on cancer label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:50:25.748408Z","iopub.execute_input":"2026-04-12T09:50:25.748744Z","iopub.status.idle":"2026-04-12T09:50:25.753400Z","shell.execute_reply.started":"2026-04-12T09:50:25.748719Z","shell.execute_reply":"2026-04-12T09:50:25.752418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img = X_val[0:1].astype(np.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:50:25.754331Z","iopub.execute_input":"2026-04-12T09:50:25.754725Z","iopub.status.idle":"2026-04-12T09:50:25.766513Z","shell.execute_reply.started":"2026-04-12T09:50:25.754691Z","shell.execute_reply":"2026-04-12T09:50:25.765849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img = img.astype(np.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:50:25.767590Z","iopub.execute_input":"2026-04-12T09:50:25.767990Z","iopub.status.idle":"2026-04-12T09:50:25.779709Z","shell.execute_reply.started":"2026-04-12T09:50:25.767946Z","shell.execute_reply":"2026-04-12T09:50:25.778799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"base_model = model.layers[0]\nlast_conv_layer = base_model.get_layer(\"out_relu\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:50:25.780802Z","iopub.execute_input":"2026-04-12T09:50:25.781102Z","iopub.status.idle":"2026-04-12T09:50:25.792385Z","shell.execute_reply.started":"2026-04-12T09:50:25.781079Z","shell.execute_reply":"2026-04-12T09:50:25.791606Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Use clean_model.input instead of model.input\ngrad_model = tf.keras.Model(\n    inputs=clean_model.input, \n    outputs=[last_conv_layer.output, clean_model.output]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:50:25.793639Z","iopub.execute_input":"2026-04-12T09:50:25.794278Z","iopub.status.idle":"2026-04-12T09:50:25.815226Z","shell.execute_reply.started":"2026-04-12T09:50:25.794230Z","shell.execute_reply":"2026-04-12T09:50:25.814412Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img = X_val[0:1].astype(\"float32\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:50:25.816623Z","iopub.execute_input":"2026-04-12T09:50:25.817256Z","iopub.status.idle":"2026-04-12T09:50:25.820987Z","shell.execute_reply.started":"2026-04-12T09:50:25.817224Z","shell.execute_reply":"2026-04-12T09:50:25.820198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\n\n# 1. Create a fresh input specifically for the Grad-CAM graph\ngrad_input = tf.keras.Input(shape=(96, 96, 3))\n\n# 2. Reconstruct the path using the layers we already have\n# We reach into the base_model to get the specific 'out_relu' output\n# but we apply it to our NEW grad_input\nx = base_model(grad_input)\nlast_conv_output = base_model.get_layer(\"out_relu\").output\n\n# 3. Connect the rest of your trained 'clean_model' layers\n# We skip the first two layers of clean_model (Input and Base) \n# and grab the head (GAP, Dense, Dropout, Dense)\ny = clean_model.layers[2](x) # GlobalAveragePooling2D\ny = clean_model.layers[3](y) # Dense 64\ny = clean_model.layers[4](y) # Dropout\nfinal_output = clean_model.layers[5](y) # Dense 1 (Sigmoid)\n\n# 4. Build the specialized Grad-CAM model\n# This model has 1 input and 2 outputs (the map and the prediction)\ngrad_model = tf.keras.Model(\n    inputs=base_model.input, \n    outputs=[base_model.get_layer(\"out_relu\").output, clean_model.output]\n)\n\n# 5. The Updated Function\ndef get_gradcam(img):\n    # Ensure image is float32 and batched (1, 96, 96, 3)\n    img = tf.cast(img, tf.float32)\n    \n    with tf.GradientTape() as tape:\n        conv_outputs, predictions = grad_model(img)\n        # For binary classification, we track the probability of the positive class\n        loss = predictions[:, 0]\n\n    # Extract gradients with respect to the last conv layer\n    grads = tape.gradient(loss, conv_outputs)\n    \n    # Pool the gradients across all axes except the channels\n    pooled_grads = tf.reduce_mean(grads, axis=(0, 1, 2))\n\n    # Weight the channels by the pooled gradients\n    conv_outputs = conv_outputs[0]\n    heatmap = conv_outputs @ pooled_grads[..., tf.newaxis]\n    \n    # Process the heatmap for visualization\n    heatmap = tf.squeeze(heatmap)\n    heatmap = tf.nn.relu(heatmap) # Only keep features that contributed positively\n    heatmap /= (tf.reduce_max(heatmap) + 1e-8)\n    \n    return heatmap.numpy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:53:15.793241Z","iopub.execute_input":"2026-04-12T09:53:15.793703Z","iopub.status.idle":"2026-04-12T09:53:15.817019Z","shell.execute_reply.started":"2026-04-12T09:53:15.793671Z","shell.execute_reply":"2026-04-12T09:53:15.816368Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Grab the base model layer from your clean_model\n# (In your code, clean_model.layers[1] is the MobileNetV2 base)\nbase_layer = clean_model.layers[1]\n\n# 2. Get the specific output of the \"out_relu\" layer from WITHIN that base layer\nlast_conv_output = base_layer.get_layer(\"out_relu\").output\n\n# 3. Reconstruct the grad_model using the verified graph path\ngrad_model = tf.keras.Model(\n    inputs=clean_model.input, \n    outputs=[last_conv_output, clean_model.output]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:51:23.304659Z","iopub.execute_input":"2026-04-12T09:51:23.305140Z","iopub.status.idle":"2026-04-12T09:51:23.321247Z","shell.execute_reply.started":"2026-04-12T09:51:23.305107Z","shell.execute_reply":"2026-04-12T09:51:23.320331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nimport matplotlib.pyplot as plt\nimport cv2\nimport numpy as np\n\n# 1. Access the base model (MobileNetV2)\nbase_layer = model.layers[0] \n\n# 2. Get the specific internal output we want to visualize\nlast_conv_output = base_layer.get_layer(\"out_relu\").output\n\n# 3. MANUALLY rebuild the prediction path from that conv output to the final result\n# This ensures Keras sees a single, unbroken \"pipe\"\nx = model.layers[1](last_conv_output) # GlobalAveragePooling2D\nx = model.layers[2](x)                # Dense (64)\nx = model.layers[3](x)                # Dropout\nfinal_output = model.layers[4](x)     # Dense (1) - The Sigmoid output\n\n# 4. Build the Grad-CAM model using this manual path\n# We use base_layer.input as the starting point\ngrad_model = tf.keras.Model(\n    inputs=base_layer.input,\n    outputs=[last_conv_output, final_output]\n)\n\n# 5. The Function (Standard Grad-CAM logic)\ndef get_gradcam(img):\n    img = tf.cast(img, tf.float32)\n    with tf.GradientTape() as tape:\n        conv_outputs, predictions = grad_model(img)\n        loss = predictions[:, 0]\n    \n    grads = tape.gradient(loss, conv_outputs)\n    pooled_grads = tf.reduce_mean(grads, axis=(0, 1, 2))\n    \n    conv_outputs = conv_outputs[0]\n    heatmap = conv_outputs @ pooled_grads[..., tf.newaxis]\n    \n    heatmap = tf.squeeze(heatmap)\n    heatmap = tf.nn.relu(heatmap)\n    heatmap /= (tf.reduce_max(heatmap) + 1e-8)\n    return heatmap.numpy()\n\n# 6. Run and Display\nimg = X_val[0:1] \nheatmap = get_gradcam(img)\n\nplt.figure(figsize=(10, 5))\nplt.subplot(1, 2, 1)\nplt.title(\"Original Image\")\nplt.imshow(X_val[0])\n\nplt.subplot(1, 2, 2)\nplt.title(\"Cancer Highlight (Grad-CAM)\")\nplt.imshow(X_val[0])\nheatmap_resized = cv2.resize(heatmap, (96, 96))\nplt.imshow(heatmap_resized, alpha=0.5, cmap='jet')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:55:46.331853Z","iopub.execute_input":"2026-04-12T09:55:46.332611Z","iopub.status.idle":"2026-04-12T09:55:47.254330Z","shell.execute_reply.started":"2026-04-12T09:55:46.332580Z","shell.execute_reply":"2026-04-12T09:55:47.253472Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in range(5):\n    img = X_val[i:i+1].astype(\"float32\")\n    heatmap = get_gradcam(img)\n\n    heatmap_resized = cv2.resize(heatmap, (96, 96))\n    heatmap_resized = np.uint8(255 * heatmap_resized)\n    heatmap_color = cv2.applyColorMap(heatmap_resized, cv2.COLORMAP_JET)\n\n    overlay = cv2.addWeighted(np.uint8(X_val[i]*255), 0.6, heatmap_color, 0.4, 0)\n\n    plt.figure()\n    plt.imshow(overlay)\n    plt.axis(\"off\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:57:09.038889Z","iopub.execute_input":"2026-04-12T09:57:09.039618Z","iopub.status.idle":"2026-04-12T09:57:10.718181Z","shell.execute_reply.started":"2026-04-12T09:57:09.039574Z","shell.execute_reply":"2026-04-12T09:57:10.717273Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras import layers\n\n# This layer will randomly flip and rotate images during training ONLY\ndata_augmentation = tf.keras.Sequential([\n  layers.RandomFlip(\"horizontal_and_vertical\"),\n  layers.RandomRotation(0.2),\n  layers.RandomZoom(0.1),\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:59:06.513219Z","iopub.execute_input":"2026-04-12T09:59:06.513777Z","iopub.status.idle":"2026-04-12T09:59:06.532608Z","shell.execute_reply.started":"2026-04-12T09:59:06.513746Z","shell.execute_reply":"2026-04-12T09:59:06.532040Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = models.Sequential([\n    layers.Input(shape=(96, 96, 3)),\n    data_augmentation, # <--- The new secret sauce\n    base_model,\n    layers.GlobalAveragePooling2D(),\n    layers.Dense(64, activation='relu'),\n    layers.Dropout(0.3),\n    layers.Dense(1, activation='sigmoid')\n])\n\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:59:15.393117Z","iopub.execute_input":"2026-04-12T09:59:15.393368Z","iopub.status.idle":"2026-04-12T09:59:15.442894Z","shell.execute_reply.started":"2026-04-12T09:59:15.393345Z","shell.execute_reply":"2026-04-12T09:59:15.442048Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = model.fit(X_train, y_train, epochs=10, validation_data=(X_val, y_val))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T09:59:22.783266Z","iopub.execute_input":"2026-04-12T09:59:22.784024Z","iopub.status.idle":"2026-04-12T09:59:56.024267Z","shell.execute_reply.started":"2026-04-12T09:59:22.783991Z","shell.execute_reply":"2026-04-12T09:59:56.023564Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nimport matplotlib.pyplot as plt\nimport cv2\nimport numpy as np\n\n# 1. Re-initialize the Grad-CAM model for the NEW augmented model\n# Based on your new summary: [Input, Augmentation, Base, GAP, Dense, Dropout, Dense]\nbase_layer_new = model.layers[1] # MobileNetV2 is now the 2nd layer\nlast_conv_output = base_layer_new.get_layer(\"out_relu\").output\n\n# Rebuild the path from the conv output to the final prediction\nx = model.layers[2](last_conv_output) # GlobalAveragePooling2D\nx = model.layers[3](x)                # Dense (64)\nx = model.layers[4](x)                # Dropout\nfinal_output = model.layers[5](x)     # Dense (1)\n\ngrad_model = tf.keras.Model(\n    inputs=base_layer_new.input,\n    outputs=[last_conv_output, final_output]\n)\n\n# 2. Grad-CAM Function\ndef get_gradcam(img):\n    img = tf.cast(img, tf.float32)\n    with tf.GradientTape() as tape:\n        conv_outputs, predictions = grad_model(img)\n        loss = predictions[:, 0]\n    \n    grads = tape.gradient(loss, conv_outputs)\n    pooled_grads = tf.reduce_mean(grads, axis=(0, 1, 2))\n    \n    conv_outputs = conv_outputs[0]\n    heatmap = conv_outputs @ pooled_grads[..., tf.newaxis]\n    \n    heatmap = tf.squeeze(heatmap)\n    heatmap = tf.nn.relu(heatmap)\n    heatmap /= (tf.reduce_max(heatmap) + 1e-8)\n    return heatmap.numpy()\n\n# 3. Visualize a few samples\nplt.figure(figsize=(15, 5))\nfor i in range(3):\n    img = X_val[i:i+1]\n    heatmap = get_gradcam(img)\n    heatmap_resized = cv2.resize(heatmap, (96, 96))\n    \n    plt.subplot(1, 3, i+1)\n    plt.imshow(X_val[i])\n    plt.imshow(heatmap_resized, alpha=0.5, cmap='jet')\n    plt.title(f\"Val Sample {i}\")\n    plt.axis(\"off\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T10:00:40.700980Z","iopub.execute_input":"2026-04-12T10:00:40.704726Z","iopub.status.idle":"2026-04-12T10:00:41.785844Z","shell.execute_reply.started":"2026-04-12T10:00:40.704691Z","shell.execute_reply":"2026-04-12T10:00:41.784705Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Update sample size and re-extract\nsample_size = 15000 \nlabels_sample = labels.sample(sample_size, random_state=42)\n\n# Load the new images\nX, y = load_images(labels_sample)\n\n# Re-split the data\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n\nprint(f\"New Training Size: {len(X_train)}\")\nprint(f\"New Validation Size: {len(X_val)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T10:01:57.833794Z","iopub.execute_input":"2026-04-12T10:01:57.834719Z","iopub.status.idle":"2026-04-12T10:03:52.124881Z","shell.execute_reply.started":"2026-04-12T10:01:57.834684Z","shell.execute_reply":"2026-04-12T10:03:52.123741Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Re-compile to reset weights for a fresh start\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n\n# Train for 10-12 epochs\nhistory = model.fit(\n    X_train, y_train, \n    epochs=12, \n    validation_data=(X_val, y_val),\n    batch_size=32\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T10:04:25.245061Z","iopub.execute_input":"2026-04-12T10:04:25.245371Z","iopub.status.idle":"2026-04-12T10:05:57.463264Z","shell.execute_reply.started":"2026-04-12T10:04:25.245347Z","shell.execute_reply":"2026-04-12T10:05:57.462490Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, classification_report\nimport seaborn as sns\n\n# Get predictions\ny_pred_final = (model.predict(X_val) > 0.5).astype(\"int32\")\n\n# Print Report\nprint(classification_report(y_val, y_pred_final, target_names=['Healthy (0)', 'Cancer (1)']))\n\n# Plot Confusion Matrix\ncm = confusion_matrix(y_val, y_pred_final)\nplt.figure(figsize=(6, 5))\nsns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Greens\")\nplt.title(\"Final Model Results\")\nplt.xlabel(\"Predicted\")\nplt.ylabel(\"Actual\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T10:06:18.331631Z","iopub.execute_input":"2026-04-12T10:06:18.335042Z","iopub.status.idle":"2026-04-12T10:06:22.927273Z","shell.execute_reply.started":"2026-04-12T10:06:18.335005Z","shell.execute_reply":"2026-04-12T10:06:22.926375Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_final_gradcam(index):\n    img = X_val[index:index+1]\n    heatmap = get_gradcam(img)\n    \n    # Resize and colorize\n    heatmap_resized = cv2.resize(heatmap, (96, 96))\n    heatmap_color = cv2.applyColorMap(np.uint8(255 * heatmap_resized), cv2.COLORMAP_JET)\n    \n    # Create a nice side-by-side\n    fig, ax = plt.subplots(1, 2, figsize=(12, 6))\n    \n    # Original\n    ax[0].imshow(X_val[index])\n    ax[0].set_title(f\"Original Slide (Label: {y_val[index]})\")\n    ax[0].axis('off')\n    \n    # Overlay\n    ax[1].imshow(X_val[index])\n    ax[1].imshow(heatmap_resized, alpha=0.4, cmap='jet')\n    ax[1].set_title(f\"Model Focus (Pred: {y_pred_final[index][0]})\")\n    ax[1].axis('off')\n    \n    plt.show()\n\n# Run it on a random sample from the new validation set\nplot_final_gradcam(np.random.randint(0, len(X_val)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T10:06:49.107710Z","iopub.execute_input":"2026-04-12T10:06:49.108633Z","iopub.status.idle":"2026-04-12T10:06:49.579365Z","shell.execute_reply.started":"2026-04-12T10:06:49.108598Z","shell.execute_reply":"2026-04-12T10:06:49.578553Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.save(\"cancer_detection_model_v1.keras\")\nprint(\"Model saved successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T10:07:02.701661Z","iopub.execute_input":"2026-04-12T10:07:02.702018Z","iopub.status.idle":"2026-04-12T10:07:03.714527Z","shell.execute_reply.started":"2026-04-12T10:07:02.701990Z","shell.execute_reply":"2026-04-12T10:07:03.713685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\ndef create_portfolio_viz(model, n_samples=4):\n    indices = np.random.choice(len(X_val), n_samples, replace=False)\n    plt.figure(figsize=(16, 8))\n    \n    for i, idx in enumerate(indices):\n        img = X_val[idx:idx+1]\n        heatmap = get_gradcam(img)\n        heatmap_resized = cv2.resize(heatmap, (96, 96))\n        \n        # Determine Label Text\n        actual = \"Cancer\" if y_val[idx] == 1 else \"Healthy\"\n        pred_prob = model.predict(img)[0][0]\n        pred = \"Cancer\" if pred_prob > 0.5 else \"Healthy\"\n        \n        plt.subplot(1, n_samples, i+1)\n        plt.imshow(X_val[idx])\n        plt.imshow(heatmap_resized, alpha=0.4, cmap='jet')\n        \n        color = 'green' if actual == pred else 'red'\n        plt.title(f\"Actual: {actual}\\nPred: {pred} ({pred_prob:.2f})\", color=color, fontsize=12)\n        plt.axis('off')\n        \n    plt.tight_layout()\n    plt.savefig(\"final_results_viz.png\", dpi=300)\n    plt.show()\n\ncreate_portfolio_viz(model)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T10:09:10.374770Z","iopub.execute_input":"2026-04-12T10:09:10.375481Z","iopub.status.idle":"2026-04-12T10:09:14.410777Z","shell.execute_reply.started":"2026-04-12T10:09:10.375447Z","shell.execute_reply":"2026-04-12T10:09:14.409736Z"}},"outputs":[],"execution_count":null}]}