{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#imports \nimport os\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nimport cv2\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, classification_report, roc_curve, auc\nfrom sklearn.utils import class_weight\nimport seaborn as sns\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models, Model, Input\nfrom tensorflow.keras.applications import MobileNetV2\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.utils import to_categorical\n\n# For reproducibility\nSEED = 42\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)\n\n# Adjust these to your environment\nCSV_PATH = \"/kaggle/input/siim-isic-melanoma-classification/train.csv\"\nJPEG_FOLDER = \"/kaggle/input/siim-isic-melanoma-classification/jpeg/train\"\nIMG_SIZE = (128, 128)     \nBATCH_SIZE = 32\nEPOCHS = 8              \nAUTOTUNE = tf.data.AUTOTUNE\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-09-30T00:33:31.194887Z","iopub.execute_input":"2025-09-30T00:33:31.195219Z","iopub.status.idle":"2025-09-30T00:33:31.203638Z","shell.execute_reply.started":"2025-09-30T00:33:31.195196Z","shell.execute_reply":"2025-09-30T00:33:31.202452Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#helper functions (load, preprocess, crop)\n\ndef read_image_cv2(path, target_size=IMG_SIZE):\n    img = cv2.imread(path)\n    if img is None:\n        return None\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = cv2.resize(img, target_size)\n    return img\n\ndef center_crop_and_resize(img, crop_frac):\n    # crop_frac = 1.0 (original), 0.75, 0.5\n    h, w = img.shape[:2]\n    ch, cw = int(h * crop_frac), int(w * crop_frac)\n    start_h = (h - ch) // 2\n    start_w = (w - cw) // 2\n    crop = img[start_h:start_h+ch, start_w:start_w+cw]\n    crop = cv2.resize(crop, IMG_SIZE)\n    return crop\n\ndef load_images_list(df, folder, img_size=IMG_SIZE, subset_limit=None):\n    images = []\n    labels = []\n    image_names = []\n    rows = df.iterrows()\n    if subset_limit:\n        rows = list(df.head(subset_limit).iterrows())\n    for _, row in tqdm(rows, total=(subset_limit or len(df))):\n        img_name = row['image_name'] + '.jpg'\n        label = int(row['target'])\n        p = os.path.join(folder, img_name)\n        if os.path.exists(p):\n            img = read_image_cv2(p, target_size=img_size)\n            if img is not None:\n                images.append(img)\n                labels.append(label)\n                image_names.append(img_name)\n    X = np.array(images, dtype=np.float32) / 255.0\n    y = np.array(labels, dtype=np.int32)\n    return X, y, image_names\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-30T00:33:35.330585Z","iopub.execute_input":"2025-09-30T00:33:35.330928Z","iopub.status.idle":"2025-09-30T00:33:35.342003Z","shell.execute_reply.started":"2025-09-30T00:33:35.330902Z","shell.execute_reply":"2025-09-30T00:33:35.340332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# load labels and sample images\ndf = pd.read_csv(CSV_PATH)\nprint(\"Total rows in CSV:\", len(df))\n\nSUBSET_LIMIT = None  \nX_all, y_all, image_names = load_images_list(df, JPEG_FOLDER, IMG_SIZE, subset_limit=SUBSET_LIMIT)\nprint(\"Loaded images:\", X_all.shape, \"Labels:\", y_all.shape)\n\n# quick distribution\nunique, counts = np.unique(y_all, return_counts=True)\nprint(\"Class distribution:\", dict(zip(unique, counts)))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-30T00:34:00.335324Z","iopub.execute_input":"2025-09-30T00:34:00.335731Z","iopub.status.idle":"2025-09-30T01:48:45.840207Z","shell.execute_reply.started":"2025-09-30T00:34:00.335706Z","shell.execute_reply":"2025-09-30T01:48:45.837690Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# create multi-scale datasets\n# We'll create three inputs per image: original (crop_frac=1.0), 0.75, 0.5\ndef make_multiscale(X_rgb):\n    X_orig = X_rgb.copy()  # images are already resized to IMG_SIZE\n    # we crop from original high-res image.\n    # We assume X_rgb is resized; emulate crops by cropping center region and resizing back.\n    X_75 = np.zeros_like(X_orig)\n    X_50 = np.zeros_like(X_orig)\n    for i in range(len(X_rgb)):\n        img = (X_rgb[i] * 255.).astype(np.uint8)\n        X_75[i] = center_crop_and_resize(img, 0.75).astype(np.float32) / 255.0\n        X_50[i] = center_crop_and_resize(img, 0.50).astype(np.float32) / 255.0\n    return X_orig, X_75, X_50\n\nX_orig, X_75, X_50 = make_multiscale((X_all * 255.).astype(np.uint8))\nprint(\"Shapes (orig,75,50):\", X_orig.shape, X_75.shape, X_50.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-30T01:48:56.847421Z","iopub.execute_input":"2025-09-30T01:48:56.848133Z","iopub.status.idle":"2025-09-30T01:49:15.975906Z","shell.execute_reply.started":"2025-09-30T01:48:56.848048Z","shell.execute_reply":"2025-09-30T01:49:15.973672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train-test split\n# We'll split maintaining stratification by y_all\ny_cat = to_categorical(y_all, num_classes=2)\n(Xo_train, Xo_test,\n X75_train, X75_test,\n X50_train, X50_test,\n y_train, y_test,\n idx_train, idx_test) = train_test_split(\n    X_orig, X_75, X_50, y_cat, np.arange(len(y_all)),\n    test_size=0.2, random_state=SEED, stratify=y_all)\n\nprint(\"Train shapes:\", Xo_train.shape, X75_train.shape, X50_train.shape, y_train.shape)\nprint(\"Test shapes:\", Xo_test.shape, X75_test.shape, X50_test.shape, y_test.shape)\n\n# Class weights (use on final combined model)\ny_train_labels = np.argmax(y_train, axis=1)\ncw = class_weight.compute_class_weight(class_weight='balanced', classes=np.unique(y_train_labels), y=y_train_labels)\nclass_weights = dict(enumerate(cw))\nprint(\"Class weights:\", class_weights)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-30T01:49:31.645275Z","iopub.execute_input":"2025-09-30T01:49:31.645668Z","iopub.status.idle":"2025-09-30T01:49:33.378277Z","shell.execute_reply.started":"2025-09-30T01:49:31.645642Z","shell.execute_reply":"2025-09-30T01:49:33.377074Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# augmentation generator\n# We'll use identical augmentation for all three inputs by augmenting indices and applying same transform.\ndatagen = ImageDataGenerator(\n    rotation_range=20,\n    width_shift_range=0.05,\n    height_shift_range=0.05,\n    zoom_range=0.1,\n    horizontal_flip=True,\n    vertical_flip=True\n)\ndatagen.fit(Xo_train)  \n\ndef multiscale_generator(Xo, X75, X50, y, batch_size=BATCH_SIZE, shuffle=True):\n    n = len(Xo)\n    indices = np.arange(n)\n    while True:\n        if shuffle:\n            np.random.shuffle(indices)\n        for start in range(0, n, batch_size):\n            end = min(start + batch_size, n)\n            batch_idx = indices[start:end]\n            # apply augmentation on each image\n            batch_o = np.zeros((len(batch_idx),) + IMG_SIZE + (3,), dtype=np.float32)\n            batch_75 = np.zeros_like(batch_o)\n            batch_50 = np.zeros_like(batch_o)\n            for i, bi in enumerate(batch_idx):\n                # augment image via datagen.random_transform\n                batch_o[i] = datagen.random_transform((Xo[bi] * 255.).astype(np.uint8)) / 255.0\n                batch_75[i] = datagen.random_transform((X75[bi] * 255.).astype(np.uint8)) / 255.0\n                batch_50[i] = datagen.random_transform((X50[bi] * 255.).astype(np.uint8)) / 255.0\n            batch_y = y[batch_idx]\n            yield [batch_o, batch_75, batch_50], batch_y\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-30T01:49:40.485547Z","iopub.execute_input":"2025-09-30T01:49:40.485973Z","iopub.status.idle":"2025-09-30T01:50:08.966715Z","shell.execute_reply.started":"2025-09-30T01:49:40.485943Z","shell.execute_reply":"2025-09-30T01:50:08.965238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\n\ndef preprocess_image(img):\n    #Step 1: Hair Removal (DullRazor approximation)\n    gray = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n    kernel = cv2.getStructuringElement(1, (17,17))\n    blackhat = cv2.morphologyEx(gray, cv2.MORPH_BLACKHAT, kernel)\n    _, mask = cv2.threshold(blackhat, 10, 255, cv2.THRESH_BINARY)\n    dst = cv2.inpaint(img, mask, 1, cv2.INPAINT_TELEA)\n\n    # Step 2: Histogram Equalization (Contrast Enhancement)\n    img_yuv = cv2.cvtColor(dst, cv2.COLOR_RGB2YUV)\n    img_yuv[:,:,0] = cv2.equalizeHist(img_yuv[:,:,0])\n    img_out = cv2.cvtColor(img_yuv, cv2.COLOR_YUV2RGB)\n\n    return img_out\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-30T01:50:24.894060Z","iopub.execute_input":"2025-09-30T01:50:24.894532Z","iopub.status.idle":"2025-09-30T01:50:24.903189Z","shell.execute_reply.started":"2025-09-30T01:50:24.894480Z","shell.execute_reply":"2025-09-30T01:50:24.901954Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers, Model, Input\nfrom tensorflow.keras.applications import MobileNetV2\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport numpy as np\n\nIMG_SIZE = (128, 128)\nBATCH_SIZE = 32\nEPOCHS = 8\n\n# Multi-scale Dataset Generator\ndef multiscale_tf_dataset(Xo, X75, X50, y, batch_size=BATCH_SIZE, shuffle=True):\n    n = len(Xo)\n    \n    def gen():\n        indices = np.arange(n)\n        while True:\n            if shuffle:\n                np.random.shuffle(indices)\n            for start in range(0, n, batch_size):\n                end = min(start + batch_size, n)\n                batch_idx = indices[start:end]\n\n                batch_o = np.array([datagen.random_transform((Xo[i]*255).astype(np.uint8))/255.0 for i in batch_idx], dtype=np.float32)\n                batch_75 = np.array([datagen.random_transform((X75[i]*255).astype(np.uint8))/255.0 for i in batch_idx], dtype=np.float32)\n                batch_50 = np.array([datagen.random_transform((X50[i]*255).astype(np.uint8))/255.0 for i in batch_idx], dtype=np.float32)\n                batch_y = np.array(y[batch_idx], dtype=np.float32)\n\n                yield (batch_o, batch_75, batch_50), batch_y\n\n    # TensorFlow output_signature\n    output_signature = (\n        (\n            tf.TensorSpec(shape=(None, *IMG_SIZE, 3), dtype=tf.float32),\n            tf.TensorSpec(shape=(None, *IMG_SIZE, 3), dtype=tf.float32),\n            tf.TensorSpec(shape=(None, *IMG_SIZE, 3), dtype=tf.float32)\n        ),\n        tf.TensorSpec(shape=(None, y.shape[1]), dtype=tf.float32)\n    )\n\n    ds = tf.data.Dataset.from_generator(gen, output_signature=output_signature)\n    ds = ds.prefetch(tf.data.AUTOTUNE)\n    return ds\n\n#Create Dataset \ntrain_ds = multiscale_tf_dataset(Xo_train, X75_train, X50_train, y_train)\nsteps_per_epoch = max(1, len(Xo_train)//BATCH_SIZE)\n\n# Build Model\nbase = MobileNetV2(include_top=False, weights=\"imagenet\", input_shape=IMG_SIZE+(3,))\nbase.trainable = False  # freeze\n\n# Inputs\nin_o = Input(shape=IMG_SIZE+(3,))\nin_75 = Input(shape=IMG_SIZE+(3,))\nin_50 = Input(shape=IMG_SIZE+(3,))\n\n# Shared base\nfeat_o   = layers.GlobalAveragePooling2D()(base(in_o))\nfeat_75  = layers.GlobalAveragePooling2D()(base(in_75))\nfeat_50  = layers.GlobalAveragePooling2D()(base(in_50))\n\nconcat = layers.concatenate([feat_o, feat_75, feat_50])\nx = layers.Dense(256, activation='relu')(concat)\nx = layers.Dropout(0.5)(x)\noutput = layers.Dense(2, activation='softmax')(x)\n\nmodel = Model(inputs=[in_o, in_75, in_50], outputs=output)\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\nmodel.summary()\n\n# Train Model\nhistory = model.fit(\n    train_ds,\n    validation_data=([Xo_test, X75_test, X50_test], y_test),\n    steps_per_epoch=steps_per_epoch,\n    epochs=EPOCHS\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-29T16:38:31.415832Z","iopub.execute_input":"2025-09-29T16:38:31.416206Z","iopub.status.idle":"2025-09-29T18:28:12.351314Z","shell.execute_reply.started":"2025-09-29T16:38:31.416176Z","shell.execute_reply":"2025-09-29T18:28:12.345922Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, classification_report, roc_curve, auc\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\n\n#Predict probabilities & labels\ny_proba = model.predict([Xo_test, X75_test, X50_test], batch_size=BATCH_SIZE, verbose=1)\ny_pred = np.argmax(y_proba, axis=1)\ny_true = np.argmax(y_test, axis=1)\n\n# Confusion Matrix\ncm = confusion_matrix(y_true, y_pred)\nprint(\"Confusion Matrix:\\n\", cm)\n\n# Plot Confusion Matrix\nplt.figure(figsize=(5,4))\nsns.heatmap(cm, annot=True, fmt='d', cmap=\"Blues\", xticklabels=[0,1], yticklabels=[0,1])\nplt.xlabel(\"Predicted\")\nplt.ylabel(\"True\")\nplt.title(\"Confusion Matrix\")\nplt.show()\n\n# Classification Report\nprint(\"\\nClassification Report:\\n\", classification_report(y_true, y_pred, digits=4, zero_division=0))\n\n# ROC & AUC\nfpr, tpr, _ = roc_curve(y_true, y_proba[:, 1])\nroc_auc = auc(fpr, tpr)\nprint(\"AUC:\", roc_auc)\n\n# Plot ROC Curve\nplt.figure(figsize=(6,5))\nplt.plot(fpr, tpr, label=f\"ROC curve (AUC = {roc_auc:.4f})\")\nplt.plot([0,1], [0,1], 'k--')  # random baseline\nplt.xlabel(\"False Positive Rate\")\nplt.ylabel(\"True Positive Rate\")\nplt.title(\"ROC Curve\")\nplt.legend(loc=\"lower right\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-29T18:42:32.710943Z","iopub.execute_input":"2025-09-29T18:42:32.711428Z","iopub.status.idle":"2025-09-29T18:44:31.231175Z","shell.execute_reply.started":"2025-09-29T18:42:32.711399Z","shell.execute_reply":"2025-09-29T18:44:31.230023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers, Model, Input\nfrom tensorflow.keras.applications import MobileNetV2\nimport tensorflow.keras.backend as K\n\n# FOCAL LOSS\ndef focal_loss(gamma=2., alpha=0.25):\n    def loss(y_true, y_pred):\n        epsilon = K.epsilon()\n        y_pred = K.clip(y_pred, epsilon, 1. - epsilon)\n        cross_entropy = -y_true * K.log(y_pred)\n        weight = alpha * K.pow((1 - y_pred), gamma)\n        loss = K.sum(weight * cross_entropy, axis=1)\n        return K.mean(loss)\n    return loss\n\n#  MODEL\nIMG_SIZE = (128, 128)\n\nbase = MobileNetV2(include_top=False, weights=\"imagenet\", input_shape=IMG_SIZE+(3,))\nbase.trainable = False  # freeze base\n\n# Inputs\nin_o = Input(shape=IMG_SIZE+(3,))\nin_75 = Input(shape=IMG_SIZE+(3,))\nin_50 = Input(shape=IMG_SIZE+(3,))\n\n# Shared base features\nfeat_o   = layers.GlobalAveragePooling2D()(base(in_o))\nfeat_75  = layers.GlobalAveragePooling2D()(base(in_75))\nfeat_50  = layers.GlobalAveragePooling2D()(base(in_50))\n\nconcat = layers.concatenate([feat_o, feat_75, feat_50])\nx = layers.Dense(256, activation='relu')(concat)\nx = layers.Dropout(0.5)(x)\noutput = layers.Dense(2, activation='softmax')(x)\n\nmodel = Model(inputs=[in_o, in_75, in_50], outputs=output)\n\n#COMPILE\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=1e-4),\n    loss=focal_loss(gamma=2., alpha=0.25),   # handles imbalance\n    metrics=['accuracy', tf.keras.metrics.Precision(name=\"precision\"), tf.keras.metrics.Recall(name=\"recall\"), tf.keras.metrics.AUC(name=\"auc\")]\n)\n\nmodel.summary()\n\n# TRAIN\nhistory = model.fit(\n    train_ds,\n    validation_data=([Xo_test, X75_test, X50_test], y_test),\n    steps_per_epoch=steps_per_epoch,\n    epochs=EPOCHS\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-29T19:18:59.523577Z","iopub.execute_input":"2025-09-29T19:18:59.524038Z","iopub.status.idle":"2025-09-29T21:01:22.302263Z","shell.execute_reply.started":"2025-09-29T19:18:59.524007Z","shell.execute_reply":"2025-09-29T21:01:22.297862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_proba = model.predict([Xo_test, X75_test, X50_test], batch_size=32)\ny_pred = np.argmax(y_proba, axis=1)\ny_true = np.argmax(y_test, axis=1)\n\nprint(\"Predicted class distribution:\", np.bincount(y_pred))\nprint(\"True class distribution:\", np.bincount(y_true))\n\n# Confusion Matrix\ncm = confusion_matrix(y_true, y_pred)\nplt.figure(figsize=(5,4))\nsns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\",\n            xticklabels=[\"Benign\",\"Melanoma\"],\n            yticklabels=[\"Benign\",\"Melanoma\"])\nplt.xlabel(\"Predicted\")\nplt.ylabel(\"True\")\nplt.title(\"Confusion Matrix\")\nplt.show()\n\n# Classification Report\nprint(\"\\nClassification Report:\\n\")\nprint(classification_report(y_true, y_pred, target_names=[\"Benign\",\"Melanoma\"], digits=4, zero_division=0))\n\n# ROC Curve & AUC\nfpr, tpr, _ = roc_curve(y_true, y_proba[:,1])\nroc_auc = auc(fpr, tpr)\n\nplt.figure(figsize=(6,5))\nplt.plot(fpr, tpr, label=f\"ROC Curve (AUC = {roc_auc:.4f})\")\nplt.plot([0,1], [0,1], 'k--')\nplt.xlabel(\"False Positive Rate\")\nplt.ylabel(\"True Positive Rate\")\nplt.title(\"ROC Curve\")\nplt.legend(loc=\"lower right\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-29T21:07:32.937108Z","iopub.execute_input":"2025-09-29T21:07:32.937552Z","iopub.status.idle":"2025-09-29T21:09:29.984919Z","shell.execute_reply.started":"2025-09-29T21:07:32.937524Z","shell.execute_reply":"2025-09-29T21:09:29.983889Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}