{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":104884,"sourceType":"datasetVersion","datasetId":54339},{"sourceId":1301322,"sourceType":"datasetVersion","datasetId":752995}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport pandas as pd\nimport pickle\nimport cv2\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint, ReduceLROnPlateau\n# from keras.callbacks import EarlyStopping, ModelCheckpoint, ReduceLROnPlateau\nfrom sklearn import svm\nfrom sklearn.metrics import  f1_score, precision_score, recall_score, accuracy_score\nimport os\nimport shutil\nimport random\nimport numpy as np\nfrom skimage.io import imread, imshow\nfrom keras.utils import img_to_array, load_img\nfrom sklearn.model_selection import train_test_split\nfrom keras.layers import concatenate, Activation, BatchNormalization, Dropout, Conv2D, Conv2DTranspose, MaxPooling2D, Input\nfrom keras.optimizers import Adam\nfrom keras.models import Model\nimport tensorflow as tf\nfrom skimage.transform import resize\nfrom skimage import io, color, feature, img_as_ubyte, measure\n\ntf.config.run_functions_eagerly(True)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Read Metadata file and copy data to train/ test folders","metadata":{}},{"cell_type":"code","source":"df_metadata = pd.read_csv('/kaggle/input/skin-cancer-mnist-ham10000/HAM10000_metadata.csv')\ntrain_data, test_data = train_test_split(df_metadata, train_size=0.75, random_state=7, stratify=df_metadata.dx)\nprint(\"success\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T09:12:30.560431Z","iopub.execute_input":"2025-12-21T09:12:30.561369Z","iopub.status.idle":"2025-12-21T09:12:30.609434Z","shell.execute_reply.started":"2025-12-21T09:12:30.561344Z","shell.execute_reply":"2025-12-21T09:12:30.608883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# fetch the values of only the \"image_id\" column form the dataframe and create a list of image names\ntrain_list = train_data['image_id'].values.tolist()\ntest_list = test_data['image_id'].values.tolist()\n\n# append .jpg and _segmentation.png values to create the image list and the segmentation mask list\ntrain_img_list = [image+'.jpg' for image in train_list]\ntest_img_list = [image+'.jpg' for image in test_list]\ntrain_seg_list = [image+'_segmentation.png' for image in train_list]\ntest_seg_list = [image+'_segmentation.png' for image in test_list]\nprint(\"success\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T09:12:45.550021Z","iopub.execute_input":"2025-12-21T09:12:45.550537Z","iopub.status.idle":"2025-12-21T09:12:45.558520Z","shell.execute_reply.started":"2025-12-21T09:12:45.550512Z","shell.execute_reply":"2025-12-21T09:12:45.557694Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Copy the masks in the test list to the data/test/masks folder\nmask_source_folder = '/kaggle/input/ham10000-lesion-segmentations/HAM10000_segmentations_lesion_tschandl'\nmask_destination_folder = 'data/test/masks'\nif not os.path.exists(mask_destination_folder):\n    os.makedirs(mask_destination_folder)\n\nfor image in test_seg_list:\n    shutil.copy(os.path.join(mask_source_folder, image), mask_destination_folder)\n    \n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Copy the test images in data/HAM10000 folder to the test folder\n# img_source_folder = '/kaggle/input/skin-cancer-mnist-ham10000'\nimg_source_folder_part_1 = '/kaggle/input/skin-cancer-mnist-ham10000/HAM10000_images_part_1'\nimg_source_folder_part_2 = '/kaggle/input/skin-cancer-mnist-ham10000/HAM10000_images_part_2'\nimg_source_folders = [img_source_folder_part_1, img_source_folder_part_2]\nimg_destination_folder = 'data/test'\n\n# for image in test_img_list:\n#     shutil.copy(os.path.join(img_source_folder, image), img_destination_folder)\nfor img_source_folder in img_source_folders:\n    for image in os.listdir(img_source_folder):\n        if image in test_img_list:\n            img_path = os.path.join(img_source_folder, image)\n            if os.path.isfile(img_path):\n                shutil.copy(img_path, img_destination_folder)\n\nprint(\"success\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T09:52:22.509358Z","iopub.execute_input":"2025-12-21T09:52:22.509898Z","iopub.status.idle":"2025-12-21T09:52:27.715507Z","shell.execute_reply.started":"2025-12-21T09:52:22.509874Z","shell.execute_reply":"2025-12-21T09:52:27.714366Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Copy the masks in the train list to the data/train/masks folder\nmask_source_folder = '/kaggle/input/ham10000-lesion-segmentations/HAM10000_segmentations_lesion_tschandl'\nmask_destination_folder = 'data/train/masks'\n# if not os.path.exists(mask_destination_folder):\n#     os.makedirs(mask_destination_folder)\n\nfor image in train_seg_list:\n    shutil.copy(os.path.join(mask_source_folder, image), mask_destination_folder)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T09:40:49.369506Z","iopub.execute_input":"2025-12-21T09:40:49.369863Z","iopub.status.idle":"2025-12-21T09:41:20.493976Z","shell.execute_reply.started":"2025-12-21T09:40:49.369840Z","shell.execute_reply":"2025-12-21T09:41:20.493105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Copy the train images in data/HAM10000 folder to the train folder\n# img_source_folder = '/kaggle/input/skin-cancer-mnist-ham10000'\nimg_source_folder_part_1 = '/kaggle/input/skin-cancer-mnist-ham10000/HAM10000_images_part_1'\nimg_source_folder_part_2 = '/kaggle/input/skin-cancer-mnist-ham10000/HAM10000_images_part_2'\nimg_source_folders = [img_source_folder_part_1, img_source_folder_part_2]\nimg_destination_folder = 'data/train'\n\n# for image in test_img_list:\n#     shutil.copy(os.path.join(img_source_folder, image), img_destination_folder)\nfor img_source_folder in img_source_folders:\n    for image in os.listdir(img_source_folder):\n        if image in train_img_list:\n            img_path = os.path.join(img_source_folder, image)\n            if os.path.isfile(img_path):\n                shutil.copy(img_path, img_destination_folder)\n\nprint(\"success\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T09:55:05.119612Z","iopub.execute_input":"2025-12-21T09:55:05.120093Z","iopub.status.idle":"2025-12-21T09:55:20.224067Z","shell.execute_reply.started":"2025-12-21T09:55:05.120068Z","shell.execute_reply":"2025-12-21T09:55:20.223453Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Create Train, Test data for Image Segmentation","metadata":{}},{"cell_type":"code","source":"IMAGE_WIDTH = 384\nIMAGE_HEIGHT = 256\nIMAGE_CHANNELS = 3","metadata":{"scrolled":true,"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T09:57:27.990517Z","iopub.execute_input":"2025-12-21T09:57:27.990808Z","iopub.status.idle":"2025-12-21T09:57:27.994739Z","shell.execute_reply.started":"2025-12-21T09:57:27.990787Z","shell.execute_reply":"2025-12-21T09:57:27.994117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_PATH = 'data/train'\nMASK_PATH = 'data/train/masks'\n\nX = np.zeros((len(train_img_list[:2000]),IMAGE_HEIGHT, IMAGE_WIDTH, IMAGE_CHANNELS), dtype=np.uint8)\nY = np.zeros((len(train_seg_list[:2000]),IMAGE_HEIGHT, IMAGE_WIDTH, 1), dtype = bool)\n\nfor n, img_name in enumerate(train_img_list[:2000]):\n    # path = DATA_PATH +'\\\\'+img_name\n    path = os.path.join(DATA_PATH, img_name)\n    img = imread(path)[:,:,:IMAGE_CHANNELS] \n    img_resize = resize(img, (IMAGE_HEIGHT,IMAGE_WIDTH,IMAGE_CHANNELS), mode = 'constant', preserve_range = True)\n    X[n] = img_resize\n    \nfor n, img_name in enumerate(train_seg_list[:2000]):\n    # path = MASK_PATH +'\\\\'+img_name\n    path = os.path.join(MASK_PATH, img_name)\n    mask = img_to_array(load_img(path, color_mode = 'grayscale'))\n    mask_resize = resize(mask, (IMAGE_HEIGHT, IMAGE_WIDTH, 1), mode = 'constant', preserve_range = True)\n    Y[n] = mask_resize","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T09:59:43.544710Z","iopub.execute_input":"2025-12-21T09:59:43.545383Z","iopub.status.idle":"2025-12-21T10:01:26.599210Z","shell.execute_reply.started":"2025-12-21T09:59:43.545356Z","shell.execute_reply":"2025-12-21T10:01:26.598435Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Perform a train, validation split from the initial train split data for training the segmentation model\nX_train, X_valid, y_train, y_valid = train_test_split(X, Y, test_size=0.2, random_state=42)\nX_train_tags, X_valid_tags, y_train_tags, y_valid_tags = train_test_split(train_img_list[:2000], train_seg_list[:2000], test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T10:02:16.609693Z","iopub.execute_input":"2025-12-21T10:02:16.610548Z","iopub.status.idle":"2025-12-21T10:02:16.852769Z","shell.execute_reply.started":"2025-12-21T10:02:16.610510Z","shell.execute_reply":"2025-12-21T10:02:16.852143Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TEST_DATA_PATH = 'data/test'\nTEST_MASK_PATH = 'data/test/masks'\n\nX_test = np.zeros((len(test_img_list[:400]),IMAGE_HEIGHT, IMAGE_WIDTH, IMAGE_CHANNELS), dtype=np.uint8)\nY_test = np.zeros((len(test_seg_list[:400]),IMAGE_HEIGHT, IMAGE_WIDTH, 1), dtype = bool)\n\nfor n, img_name in enumerate(test_img_list[:400]):\n    # path = TEST_DATA_PATH +'\\\\'+img_name\n    path = os.path.join(TEST_DATA_PATH, img_name)\n    img = imread(path)[:,:,:IMAGE_CHANNELS] \n    img_resize = resize(img, (IMAGE_HEIGHT,IMAGE_WIDTH,IMAGE_CHANNELS), mode = 'constant', preserve_range = True)\n    X_test[n] = img_resize\n\nfor n, img_name in enumerate(test_seg_list[:400]):\n    # path = TEST_MASK_PATH +'\\\\'+img_name\n    path = os.path.join(TEST_MASK_PATH, img_name)\n    mask = img_to_array(load_img(path, color_mode = 'grayscale'))\n    mask_resize = resize(mask, (IMAGE_HEIGHT, IMAGE_WIDTH, 1), mode = 'constant', preserve_range = True)\n    Y_test[n] = mask_resize\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T10:04:41.489924Z","iopub.execute_input":"2025-12-21T10:04:41.490451Z","iopub.status.idle":"2025-12-21T10:05:02.121759Z","shell.execute_reply.started":"2025-12-21T10:04:41.490427Z","shell.execute_reply":"2025-12-21T10:05:02.121139Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualize the X_train and y_train data","metadata":{}},{"cell_type":"code","source":"index = random.randint(0, len(X_train))\nimshow(X_train[index])","metadata":{"scrolled":true,"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T10:07:24.001757Z","iopub.execute_input":"2025-12-21T10:07:24.002303Z","iopub.status.idle":"2025-12-21T10:07:24.295484Z","shell.execute_reply.started":"2025-12-21T10:07:24.002278Z","shell.execute_reply":"2025-12-21T10:07:24.294655Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"imshow(np.squeeze(y_train[index].astype(np.float32)))","metadata":{"scrolled":true,"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T10:07:38.453198Z","iopub.execute_input":"2025-12-21T10:07:38.453936Z","iopub.status.idle":"2025-12-21T10:07:38.671138Z","shell.execute_reply.started":"2025-12-21T10:07:38.453899Z","shell.execute_reply":"2025-12-21T10:07:38.670302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(X_train_tags[index], y_train_tags[index] )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T10:07:43.703704Z","iopub.execute_input":"2025-12-21T10:07:43.704211Z","iopub.status.idle":"2025-12-21T10:07:43.708145Z","shell.execute_reply.started":"2025-12-21T10:07:43.704163Z","shell.execute_reply":"2025-12-21T10:07:43.707435Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Define the UNET Model for Image Segmentation","metadata":{}},{"cell_type":"code","source":"def get_conv2d_layers(input_tensor, n_filters, kernel_size = 3, batchnorm = True):\n    \"\"\"Function to create convolution layers with the given input parameters\"\"\"\n    # Layer 1\n    x = Conv2D(filters = n_filters, kernel_size = (kernel_size, kernel_size),\\\n              kernel_initializer = 'he_normal', padding = 'same')(input_tensor)\n    if batchnorm:\n        x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    \n    # Layer 2\n    x = Conv2D(filters = n_filters, kernel_size = (kernel_size, kernel_size),\\\n              kernel_initializer = 'he_normal', padding = 'same')(input_tensor)\n    if batchnorm:\n        x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    return x\n\ndef unet(input_img, n_filters = 16, dropout = 0.1, batchnorm = True):\n    \"\"\"Function to define the UNET Model\"\"\"\n    # UNET Contracting path\n    conv1 = get_conv2d_layers(input_img, n_filters * 1, kernel_size = 3, batchnorm = batchnorm)\n    pool1 = MaxPooling2D((2, 2))(conv1)\n    pool1 = Dropout(dropout)(pool1)\n    \n    conv2 = get_conv2d_layers(pool1, n_filters * 2, kernel_size = 3, batchnorm = batchnorm)\n    pool2 = MaxPooling2D((2, 2))(conv2)\n    pool2 = Dropout(dropout)(pool2)\n    \n    conv3 = get_conv2d_layers(pool2, n_filters * 4, kernel_size = 3, batchnorm = batchnorm)\n    pool3 = MaxPooling2D((2, 2))(conv3)\n    pool3 = Dropout(dropout)(pool3)\n    \n    conv4 = get_conv2d_layers(pool3, n_filters * 8, kernel_size = 3, batchnorm = batchnorm)\n    pool4 = MaxPooling2D((2, 2))(conv4)\n    pool4 = Dropout(dropout)(pool4)\n    \n    conv5 = get_conv2d_layers(pool4, n_filters = n_filters * 16, kernel_size = 3, batchnorm = batchnorm)\n    \n    # UNET Expansive path\n    up6 = Conv2DTranspose(n_filters * 8, (3, 3), strides = (2, 2), padding = 'same')(conv5)\n    up6 = concatenate([up6, conv4])\n    up6 = Dropout(dropout)(up6)\n    conv6 = get_conv2d_layers(up6, n_filters * 8, kernel_size = 3, batchnorm = batchnorm)\n    \n    up7 = Conv2DTranspose(n_filters * 4, (3, 3), strides = (2, 2), padding = 'same')(conv6)\n    up7 = concatenate([up7, conv3])\n    up7 = Dropout(dropout)(up7)\n    conv7 = get_conv2d_layers(up7, n_filters * 4, kernel_size = 3, batchnorm = batchnorm)\n    \n    up8 = Conv2DTranspose(n_filters * 2, (3, 3), strides = (2, 2), padding = 'same')(conv7)\n    up8 = concatenate([up8, conv2])\n    up8 = Dropout(dropout)(up8)\n    conv8 = get_conv2d_layers(up8, n_filters * 2, kernel_size = 3, batchnorm = batchnorm)\n    \n    up9 = Conv2DTranspose(n_filters * 1, (3, 3), strides = (2, 2), padding = 'same')(conv8)\n    up9 = concatenate([up9, conv1])\n    up9 = Dropout(dropout)(up9)\n    conv9 = get_conv2d_layers(up9, n_filters * 1, kernel_size = 3, batchnorm = batchnorm)\n    \n    outputs = Conv2D(1, (1, 1), activation='sigmoid')(conv9)\n    model = Model(inputs=[input_img], outputs=[outputs])\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T10:07:52.907245Z","iopub.execute_input":"2025-12-21T10:07:52.907713Z","iopub.status.idle":"2025-12-21T10:07:52.917878Z","shell.execute_reply.started":"2025-12-21T10:07:52.907686Z","shell.execute_reply":"2025-12-21T10:07:52.917264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# intialize the model and compile it with giving as inputs the optimizer type, loss function and metrics type\nimage_dims = Input((IMAGE_HEIGHT,IMAGE_WIDTH, IMAGE_CHANNELS), name='img')\nmodel = unet(image_dims, n_filters=16, dropout=0.05, batchnorm=True)\nmodel.compile(optimizer=Adam(), loss=\"binary_crossentropy\", metrics=[\"accuracy\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T10:07:57.564005Z","iopub.execute_input":"2025-12-21T10:07:57.564570Z","iopub.status.idle":"2025-12-21T10:07:59.795864Z","shell.execute_reply.started":"2025-12-21T10:07:57.564545Z","shell.execute_reply":"2025-12-21T10:07:59.795214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# view the summary of the created UNET model\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T10:08:28.301574Z","iopub.execute_input":"2025-12-21T10:08:28.301849Z","iopub.status.idle":"2025-12-21T10:08:28.352510Z","shell.execute_reply.started":"2025-12-21T10:08:28.301828Z","shell.execute_reply":"2025-12-21T10:08:28.351920Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"callbacks = [\n    EarlyStopping(patience=10, verbose=1),\n    ReduceLROnPlateau(factor=0.1, patience=5, min_lr=0.00001, verbose=1),\n    ModelCheckpoint('model-skin-lesion-segmentation-org2000.h5', verbose=1, save_best_only=True, save_weights_only=False)\n]\nprint(\"success\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T10:10:50.255318Z","iopub.execute_input":"2025-12-21T10:10:50.256090Z","iopub.status.idle":"2025-12-21T10:10:50.260459Z","shell.execute_reply.started":"2025-12-21T10:10:50.256066Z","shell.execute_reply":"2025-12-21T10:10:50.259817Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train the UNET Model","metadata":{}},{"cell_type":"code","source":"# Below is the code to train the defined UNET model for 10 epochs with a batch size of 32.\n# The training has been stopped at epoch 8 as there is no further improvement in the validation loss post epoch 5\nresults = model.fit(X_train, y_train, batch_size=32, epochs=10, callbacks=callbacks,\\\n                    validation_data=(X_valid, y_valid))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T10:10:53.051106Z","iopub.execute_input":"2025-12-21T10:10:53.051781Z","iopub.status.idle":"2025-12-21T10:17:34.214278Z","shell.execute_reply.started":"2025-12-21T10:10:53.051756Z","shell.execute_reply":"2025-12-21T10:17:34.213679Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Once the model training has been done, from next time onwards we can just load the model weights instead of retraining the model\nmodel.load_weights('model-skin-lesion-segmentation-org2000.h5')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T10:18:36.017780Z","iopub.execute_input":"2025-12-21T10:18:36.018040Z","iopub.status.idle":"2025-12-21T10:18:36.092865Z","shell.execute_reply.started":"2025-12-21T10:18:36.018022Z","shell.execute_reply":"2025-12-21T10:18:36.092149Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.evaluate(X_test, Y_test, verbose=1)","metadata":{"scrolled":true,"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T10:18:39.971424Z","iopub.execute_input":"2025-12-21T10:18:39.971688Z","iopub.status.idle":"2025-12-21T10:18:42.252955Z","shell.execute_reply.started":"2025-12-21T10:18:39.971669Z","shell.execute_reply":"2025-12-21T10:18:42.252417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predicted = model.predict(X_test, verbose=1)\npredicted = (predicted > 0.5).astype(bool)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T10:18:54.964432Z","iopub.execute_input":"2025-12-21T10:18:54.965029Z","iopub.status.idle":"2025-12-21T10:18:57.435871Z","shell.execute_reply.started":"2025-12-21T10:18:54.965003Z","shell.execute_reply":"2025-12-21T10:18:57.435291Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Plot the predicted UNET Masks","metadata":{}},{"cell_type":"code","source":"def plot_masks(X_test, Y_test, predicted, idx=None):\n    \"\"\"Function to visualize the masks predicted by the UNET model\"\"\"\n\n    figure, ax = plt.subplots(1, 3, figsize=(20, 10))\n    if idx is None:\n        idx = random.randint(0, len(X_test))\n\n    ax[0].imshow(X_test[idx])\n    ax[0].contour(np.squeeze(predicted[idx]))\n    ax[0].set_title('Original Image')\n\n    ax[1].imshow(Y_test[idx].squeeze())\n    ax[1].set_title('Ground Truth Mask')\n\n    ax[2].imshow(predicted[idx].squeeze())\n    ax[2].set_title('UNET Predicted Mask')\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T10:19:03.430885Z","iopub.execute_input":"2025-12-21T10:19:03.431149Z","iopub.status.idle":"2025-12-21T10:19:03.436444Z","shell.execute_reply.started":"2025-12-21T10:19:03.431131Z","shell.execute_reply":"2025-12-21T10:19:03.435615Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_masks(X_test, Y_test, predicted, idx= 190)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T10:19:06.708489Z","iopub.execute_input":"2025-12-21T10:19:06.708984Z","iopub.status.idle":"2025-12-21T10:19:07.339720Z","shell.execute_reply.started":"2025-12-21T10:19:06.708958Z","shell.execute_reply":"2025-12-21T10:19:07.339008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from skimage.io import imsave\noutput_image_dir = 'output/predicted_masks_images'\noutput_mask_dir = 'output/predicted_masks_npy'\n# output_4channel_dir = 'predicted_4channel_images'\n\n\nos.makedirs(output_image_dir, exist_ok=True)\nos.makedirs(output_mask_dir, exist_ok=True)\n# os.makedirs(output_4channel_dir, exist_ok=True)\n\nfor idx in range(len(predicted)):\n    # original_image = X_test[idx]\n    predicted_mask = predicted[idx].squeeze()\n\n    save_path_image = os.path.join(output_image_dir, f\"predicted_mask_{idx}.png\")\n    imsave(save_path_image, predicted_mask.astype(np.uint8) * 255)\n\n    save_path_mask = os.path.join(output_mask_dir, f\"predicted_mask_{idx}.npy\")\n    np.save(save_path_mask, predicted_mask.astype(np.uint8)) \n\nprint(\"success\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T10:33:04.415915Z","iopub.execute_input":"2025-12-21T10:33:04.416480Z","iopub.status.idle":"2025-12-21T10:33:05.586664Z","shell.execute_reply.started":"2025-12-21T10:33:04.416454Z","shell.execute_reply":"2025-12-21T10:33:05.585834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_masks(X_test, Y_test, predicted, idx= 120)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T10:34:32.644993Z","iopub.execute_input":"2025-12-21T10:34:32.645528Z","iopub.status.idle":"2025-12-21T10:34:33.267888Z","shell.execute_reply.started":"2025-12-21T10:34:32.645501Z","shell.execute_reply":"2025-12-21T10:34:33.267143Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_masks(X_test, Y_test, predicted, idx= 15)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_masks(X_test, Y_test, predicted, idx= 33)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_masks(X_test, Y_test, predicted, idx= 351)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_masks(X_test, Y_test, predicted, idx= 350)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_masks(X_test, Y_test, predicted, idx= 4)\n# We see that the trained model does not perform very well when there's lot of noise in the \n# image (hair, gel blobs, water droplets) so, we later on perform preprocessing on images before sending them to our model.","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_masks(X_test, Y_test, predicted, idx= 0)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Extract Features from Segmentation Masks to perform Tumor classification","metadata":{}},{"cell_type":"markdown","source":"### Create Train, Test data for Image classification","metadata":{}},{"cell_type":"code","source":"# From the segmentation masks in the test data create the test data for classification\ntest_dict = {}\nfor i in range(Y_test.shape[0]):\n    img_label = measure.label(np.squeeze(Y_test[i])) \n    region_props = measure.regionprops(img_label)\n    num_labels = len(region_props)\n    areas = [region.area for region in region_props]\n    if num_labels > 0 and areas[np.argmax(areas)] >= 1200:\n        target_label = region_props[np.argmax(areas)].label\n        test_dict[test_img_list[i]] = region_props[target_label -1]\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# From the segmentation masks in the train data create the train data for classification\ntrain_dict = {}\nfor i in range(y_train.shape[0]):\n    img_label_train = measure.label(np.squeeze(y_train[i])) \n    props_train = measure.regionprops(img_label_train)\n    num_labels = len(props_train)\n    areas = [region.area for region in props_train]\n    if num_labels > 0 and areas[np.argmax(areas)] >= 1200:\n        target_label = props_train[np.argmax(areas)].label\n        train_dict[X_train_tags[i]] = props_train[target_label -1]","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(train_dict)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(test_dict)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Define Function to extract the ABCDT features from the segmentation masks","metadata":{}},{"cell_type":"code","source":"# function code referenced from https://github.com/biagiom/skin-lesions-classifier/blob/master/skin_lesions_classifier.ipynb\ndef extract_features(input_dict, base_path):\n    \"\"\"Function to perform feature extraction for the segmented masks by calculating their Asymmetry, Border irregularity,\n    Color variation, Diameter and Texture\"\"\"\n    features = {}\n    for idx, image_name in enumerate(input_dict):\n        path = base_path+'\\\\'+image_name\n        image = imread(path)\n        gray_img = color.rgb2gray(image)\n        lesion_region = input_dict[image_name]\n\n        # Asymmetry\n        area_total = lesion_region.area\n        img_mask = lesion_region.image\n        horizontal_flip = np.fliplr(img_mask)\n        diff_horizontal = img_mask * ~horizontal_flip\n        vertical_flip = np.flipud(img_mask)\n        diff_vertical = img_mask * ~vertical_flip\n        diff_horizontal_area = np.count_nonzero(diff_horizontal)\n        diff_vertical_area = np.count_nonzero(diff_vertical)\n        asymm_idx = 0.5 * ((diff_horizontal_area / area_total) + (diff_vertical_area / area_total))\n        ecc = lesion_region.eccentricity\n\n        # Border irregularity\n        compact_index = (lesion_region.perimeter ** 2) / (4 * np.pi * area_total)\n\n\n        # Color variegation:\n        sliced = image[lesion_region.slice]\n        lesion_r = sliced[:, :, 0]\n        lesion_g = sliced[:, :, 1]\n        lesion_b = sliced[:, :, 2]\n        C_r = np.std(lesion_r) / np.max(lesion_r)\n        C_g = np.std(lesion_g) / np.max(lesion_g)\n        C_b = np.std(lesion_b) / np.max(lesion_b)\n\n        # Diameter:\n        eq_diameter = lesion_region.equivalent_diameter\n\n\n        # Texture:\n        glcm = feature.greycomatrix(image=img_as_ubyte(gray_img), distances=[1],\n                                    angles=[0, np.pi/4, np.pi/2, np.pi * 3/2],\n                                    symmetric=True, normed=True)\n        correlation = np.mean(feature.greycoprops(glcm, prop='correlation'))\n        homogeneity = np.mean(feature.greycoprops(glcm, prop='homogeneity'))\n        energy = np.mean(feature.greycoprops(glcm, prop='energy'))\n        contrast = np.mean(feature.greycoprops(glcm, prop='contrast'))\n\n        features[image_name] = [asymm_idx, ecc, compact_index, C_r, C_g, C_b,\n                                       eq_diameter, correlation, homogeneity, energy, contrast]\n    return features","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Invoke the extract_features function to calculate the ABCDT properties for train, test data\ntrain_features = extract_features(train_dict, 'data/train')\ntest_features = extract_features(test_dict, 'data/test')","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in range(df_metadata['image_id'].count()):\n    df_metadata.at[i, 'image_id'] += '.jpg'\n    \n# fetch the target variable 'dx' for the train, test data\ntest_df = df_metadata.loc[df_metadata['image_id'].isin(list(test_features.keys()))]\ntrain_df = df_metadata.loc[df_metadata['image_id'].isin(list(train_features.keys()))]","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df","metadata":{"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert values to list\nX_train_svm = list(train_features.values())\ny_train_svm = train_df['dx'].values.tolist()\n\nX_test_svm = list(test_features.values())\ny_test_svm = test_df['dx'].values.tolist()","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert the cancer classes in target variable as cancerous or non-cancerous\nfor i in range(len(y_train_svm)):\n    if y_train_svm[i] in ['bkl','df','nv','vasc']:\n        y_train_svm[i] = 'benign'\n    else:\n        y_train_svm[i] = 'cancerous'\n        \nfor i in range(len(y_test_svm)):\n    if y_test_svm[i] in ['bkl','df','nv','vasc']:\n        y_test_svm[i] = 'benign'\n    else:\n        y_test_svm[i] = 'cancerous'","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Train SVM Model to perform Cancer Classification","metadata":{}},{"cell_type":"code","source":"# Train the SVM polynomial kernel to perform classification on our dataset\nclf = svm.SVC(gamma=0.001, kernel='poly', degree=5)\nclf = clf.fit(X_train_svm, y_train_svm)\n\ny_pred = clf.predict(X_test_svm)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# save the SVM classification model to .pkl file for later use\nwith open('svm_model_poly.pkl', 'wb') as file:\n    pickle.dump(clf, file)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### SVM Performance metrics","metadata":{}},{"cell_type":"code","source":"# Calculate the performance metrics of the SVM Classifier\naccuracy = accuracy_score(y_test_svm, y_pred)\nf1 = f1_score(y_test_svm, y_pred, average='weighted')\nrecall = recall_score(y_test_svm, y_pred, average='weighted')\nprecision = precision_score(y_test_svm, y_pred, average='weighted')\n\nprint('Accuracy: {:.3f}'.format(accuracy))\nprint('F1-score: {:.3f}'.format(f1))\nprint('Recall: {:.3f}'.format(recall))\nprint('Precision: {:.3f}'.format(precision))","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Function to Pre-Process images to be used for segmentation","metadata":{}},{"cell_type":"code","source":"# The steps in the below function are being used to perform image pre-processing using the PyQT terminal\ndef pre_process_images():\n    \"\"\"Function to perform removal of hair in images by using the dull razor method\"\"\"\n    \n    image_path = 'data/test'\n    abs_path = os.path.abspath('data/test/preprocessed')\n\n    for img_name in test_image_ids[:20]:\n        full_path = os.path.join(image_path, img_name)\n        preprocess_path = os.path.join(abs_path, img_name)\n\n        image = io.imread(full_path)\n        # Convert the image to grayscale\n        grayScale = cv2.cvtColor(image, cv2.COLOR_RGB2GRAY)\n\n        # Define kernel to perform Morphological operations\n        kernel = cv2.getStructuringElement(1, (17, 17))\n\n        # Filter the grayscale image using Blackhat filtering technique to find the hair contours\n        blackhat = cv2.morphologyEx(grayScale, cv2.MORPH_BLACKHAT, kernel)\n\n        # Increase the intensity of the hair contours to perform inpainting of the image\n        ret, thresh2 = cv2.threshold(blackhat, 10, 255, cv2.THRESH_BINARY)\n\n        # Perform inpainting of the original image using the thresholds calculated above\n        dst = cv2.inpaint(image, thresh2, 1, cv2.INPAINT_TELEA)\n        img_filtered = cv2.GaussianBlur(dst,(7,7),0)\n        io.imsave(preprocess_path, img_filtered)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pre_process_images()","metadata":{},"outputs":[],"execution_count":null}]}