{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"}],"dockerImageVersionId":30299,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    print(os.path.join(dirname))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:47:32.310263Z","iopub.execute_input":"2025-10-12T08:47:32.310633Z","iopub.status.idle":"2025-10-12T08:49:11.195668Z","shell.execute_reply.started":"2025-10-12T08:47:32.310544Z","shell.execute_reply":"2025-10-12T08:49:11.194710Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import re\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly\nimport plotly.graph_objects as go\nimport cv2\nimport tensorflow as tf\nfrom kaggle_datasets import KaggleDatasets\nfrom functools import partial\nimport sklearn\nfrom tqdm import tqdm_notebook as tqdm\nimport gc\n%matplotlib inline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:11.197505Z","iopub.execute_input":"2025-10-12T08:49:11.197832Z","iopub.status.idle":"2025-10-12T08:49:16.227400Z","shell.execute_reply.started":"2025-10-12T08:49:11.197801Z","shell.execute_reply":"2025-10-12T08:49:16.226336Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"try:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nexcept:\n    strategy = tf.distribute.get_strategy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:16.228631Z","iopub.execute_input":"2025-10-12T08:49:16.229083Z","iopub.status.idle":"2025-10-12T08:49:16.238838Z","shell.execute_reply.started":"2025-10-12T08:49:16.229058Z","shell.execute_reply":"2025-10-12T08:49:16.238116Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('Number of replicas:', strategy.num_replicas_in_sync)\nprint(\"Version of Tensorflow used : \", tf.__version__)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:16.239944Z","iopub.execute_input":"2025-10-12T08:49:16.240374Z","iopub.status.idle":"2025-10-12T08:49:16.246629Z","shell.execute_reply.started":"2025-10-12T08:49:16.240341Z","shell.execute_reply":"2025-10-12T08:49:16.245647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"AUTOTUNE = tf.data.experimental.AUTOTUNE\nGCS_PATH = KaggleDatasets().get_gcs_path()\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nIMAGE_SIZE = [1024, 1024]\nSHAPE = [256, 256]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:16.247879Z","iopub.execute_input":"2025-10-12T08:49:16.248300Z","iopub.status.idle":"2025-10-12T08:49:16.510376Z","shell.execute_reply.started":"2025-10-12T08:49:16.248266Z","shell.execute_reply":"2025-10-12T08:49:16.509501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Batch Size = \", BATCH_SIZE)\nprint(\"GCS Path = \", GCS_PATH)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:16.511494Z","iopub.execute_input":"2025-10-12T08:49:16.511836Z","iopub.status.idle":"2025-10-12T08:49:16.516982Z","shell.execute_reply.started":"2025-10-12T08:49:16.511802Z","shell.execute_reply":"2025-10-12T08:49:16.516020Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.DataFrame(pd.read_csv(\"../input/siim-isic-melanoma-classification/train.csv\"))\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:16.520318Z","iopub.execute_input":"2025-10-12T08:49:16.520608Z","iopub.status.idle":"2025-10-12T08:49:16.614040Z","shell.execute_reply.started":"2025-10-12T08:49:16.520587Z","shell.execute_reply":"2025-10-12T08:49:16.613021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = pd.DataFrame(pd.read_csv(\"../input/siim-isic-melanoma-classification/test.csv\"))\ntest.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:16.615221Z","iopub.execute_input":"2025-10-12T08:49:16.615554Z","iopub.status.idle":"2025-10-12T08:49:16.648562Z","shell.execute_reply.started":"2025-10-12T08:49:16.615526Z","shell.execute_reply":"2025-10-12T08:49:16.647799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:16.649902Z","iopub.execute_input":"2025-10-12T08:49:16.650570Z","iopub.status.idle":"2025-10-12T08:49:16.679875Z","shell.execute_reply.started":"2025-10-12T08:49:16.650536Z","shell.execute_reply":"2025-10-12T08:49:16.679049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:16.680812Z","iopub.execute_input":"2025-10-12T08:49:16.681080Z","iopub.status.idle":"2025-10-12T08:49:16.693265Z","shell.execute_reply.started":"2025-10-12T08:49:16.681056Z","shell.execute_reply":"2025-10-12T08:49:16.692312Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dir = \"/kaggle/input/siim-isic-melanoma-classification/jpeg/train/\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:16.694296Z","iopub.execute_input":"2025-10-12T08:49:16.694527Z","iopub.status.idle":"2025-10-12T08:49:16.699304Z","shell.execute_reply.started":"2025-10-12T08:49:16.694507Z","shell.execute_reply":"2025-10-12T08:49:16.698489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_names = train[\"image_name\"].values + \".jpg\"\nrandom_images = [np.random.choice(image_names) for i in range(4)] # Generates a random sample from a given 1-D array\nrandom_images ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:16.700539Z","iopub.execute_input":"2025-10-12T08:49:16.701363Z","iopub.status.idle":"2025-10-12T08:49:16.713070Z","shell.execute_reply.started":"2025-10-12T08:49:16.701326Z","shell.execute_reply":"2025-10-12T08:49:16.712139Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_images = []","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:16.714199Z","iopub.execute_input":"2025-10-12T08:49:16.714455Z","iopub.status.idle":"2025-10-12T08:49:16.719439Z","shell.execute_reply.started":"2025-10-12T08:49:16.714416Z","shell.execute_reply":"2025-10-12T08:49:16.718546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (12, 8))\nfor i in range(4) : \n    plt.subplot(2, 2, i + 1) \n    image = cv2.imread(os.path.join(train_dir, random_images[i]))\n    # cv2 reads images in BGR format. Hence we convert it to RGB\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    sample_images.append(image)\n    plt.imshow(image, cmap = \"gray\")\n    plt.grid(True)\n# Automatically adjust subplot parameters to give specified padding.\nplt.tight_layout()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:16.720630Z","iopub.execute_input":"2025-10-12T08:49:16.721165Z","iopub.status.idle":"2025-10-12T08:49:26.170600Z","shell.execute_reply.started":"2025-10-12T08:49:16.721141Z","shell.execute_reply":"2025-10-12T08:49:26.169628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def non_local_means_denoising(image) : \n    denoised_image = cv2.fastNlMeansDenoisingColored(image, None, 10, 10, 7, 21)\n    return denoised_image","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:26.171803Z","iopub.execute_input":"2025-10-12T08:49:26.172117Z","iopub.status.idle":"2025-10-12T08:49:26.177063Z","shell.execute_reply.started":"2025-10-12T08:49:26.172087Z","shell.execute_reply":"2025-10-12T08:49:26.176237Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_image = cv2.imread(os.path.join(train_dir, random_images[0]))\n# cv2 reads images in BGR format. Hence we convert it to RGB\nsample_image = cv2.cvtColor(sample_image, cv2.COLOR_BGR2RGB)\ndenoised_image = non_local_means_denoising(sample_image)\n\n\nplt.figure(figsize = (12, 8))\nplt.subplot(1,2,1)\nplt.imshow(sample_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"Normal Image\")\n\nplt.subplot(1,2,2)  \nplt.imshow(denoised_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"Denoised image\")    \n# Automatically adjust subplot parameters to give specified padding.\nplt.tight_layout() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:26.178527Z","iopub.execute_input":"2025-10-12T08:49:26.178822Z","iopub.status.idle":"2025-10-12T08:49:27.381217Z","shell.execute_reply.started":"2025-10-12T08:49:26.178797Z","shell.execute_reply":"2025-10-12T08:49:27.380402Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def histogram_equalization(image) : \n    image_ycrcb = cv2.cvtColor(image, cv2.COLOR_RGB2YCR_CB)\n    y_channel = image_ycrcb[:,:,0] # apply local histogram processing on this channel\n    cr_channel = image_ycrcb[:,:,1]\n    cb_channel = image_ycrcb[:,:,2]\n    \n    # Local histogram equalization\n    clahe = cv2.createCLAHE(clipLimit = 2.0, tileGridSize=(8,8))\n    equalized = clahe.apply(y_channel)\n    equalized_image = cv2.merge([equalized, cr_channel, cb_channel])\n    equalized_image = cv2.cvtColor(equalized_image, cv2.COLOR_YCR_CB2RGB)\n    return equalized_image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:27.382559Z","iopub.execute_input":"2025-10-12T08:49:27.382971Z","iopub.status.idle":"2025-10-12T08:49:27.390786Z","shell.execute_reply.started":"2025-10-12T08:49:27.382935Z","shell.execute_reply":"2025-10-12T08:49:27.389792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"equalized_image = histogram_equalization(denoised_image)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:27.391897Z","iopub.execute_input":"2025-10-12T08:49:27.392184Z","iopub.status.idle":"2025-10-12T08:49:27.413327Z","shell.execute_reply.started":"2025-10-12T08:49:27.392157Z","shell.execute_reply":"2025-10-12T08:49:27.412425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (12, 8))\nplt.subplot(1,3,1)\nplt.imshow(sample_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"Normal Image\", fontsize = 14)\n\nplt.subplot(1,3,2)  \nplt.imshow(denoised_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"denoised image after histogram processing\", fontsize = 14)\n\nplt.subplot(1,3,3)  \nplt.imshow(equalized_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"Histogram equalized image\", fontsize = 14)\n# Automatically adjust subplot parameters to give specified padding.\nplt.tight_layout()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:27.414486Z","iopub.execute_input":"2025-10-12T08:49:27.414763Z","iopub.status.idle":"2025-10-12T08:49:28.092993Z","shell.execute_reply.started":"2025-10-12T08:49:27.414714Z","shell.execute_reply":"2025-10-12T08:49:28.092120Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def segmentation(image, k, attempts) : \n    vectorized = np.float32(image.reshape((-1, 3)))\n    criteria = (cv2.TERM_CRITERIA_EPS + cv2.TERM_CRITERIA_MAX_ITER, 20, 1.0)\n    res , label , center = cv2.kmeans(vectorized, k, None, criteria, attempts, cv2.KMEANS_PP_CENTERS)\n    center = np.uint8(center)\n    res = center[label.flatten()]\n    segmented_image = res.reshape((image.shape))\n    return segmented_image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:28.094234Z","iopub.execute_input":"2025-10-12T08:49:28.094573Z","iopub.status.idle":"2025-10-12T08:49:28.101439Z","shell.execute_reply.started":"2025-10-12T08:49:28.094530Z","shell.execute_reply":"2025-10-12T08:49:28.100457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (12, 8))\nplt.subplot(1,1,1)\nplt.imshow(denoised_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"de Noised Image\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:28.102581Z","iopub.execute_input":"2025-10-12T08:49:28.102885Z","iopub.status.idle":"2025-10-12T08:49:28.457759Z","shell.execute_reply.started":"2025-10-12T08:49:28.102860Z","shell.execute_reply":"2025-10-12T08:49:28.456918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (12, 8))\nsegmented_image = segmentation(denoised_image, 3, 10) # k = 3, attempt = 10\nplt.subplot(1,3,1)\nplt.imshow(segmented_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"Segmented Image with k = 3\")\n\nsegmented_image = segmentation(denoised_image, 4, 10) # k = 4, attempt = 10\nplt.subplot(1,3,2)\nplt.imshow(segmented_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"Segmented Image with k = 4\")\n\nsegmented_image = segmentation(denoised_image, 5, 10) # k = 5, attempt = 10\nplt.subplot(1,3,3)\nplt.imshow(segmented_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"Segmented Image with k = 5\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:28.466197Z","iopub.execute_input":"2025-10-12T08:49:28.466534Z","iopub.status.idle":"2025-10-12T08:49:30.831618Z","shell.execute_reply.started":"2025-10-12T08:49:28.466504Z","shell.execute_reply":"2025-10-12T08:49:30.830770Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split \ntraining_files, validation_files = train_test_split(tf.io.gfile.glob(GCS_PATH + \"/tfrecords/train*.tfrec\"),\n                                                   test_size = 0.1, random_state = 42)\n\ntesting_files = tf.io.gfile.glob(GCS_PATH + \"/tfrecords/test*.tfrec\")\n\nprint(\"Number of training files = \", len(training_files))\nprint(\"Number of validation files = \", len(validation_files))\nprint(\"Number of test files = \", len(testing_files))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:30.833041Z","iopub.execute_input":"2025-10-12T08:49:30.833792Z","iopub.status.idle":"2025-10-12T08:49:31.122080Z","shell.execute_reply.started":"2025-10-12T08:49:30.833753Z","shell.execute_reply":"2025-10-12T08:49:31.120675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def decode_image(image) : \n    image = tf.image.decode_jpeg(image, channels = 3)\n    image = tf.cast(image, tf.float32)\n    image = image / 255.0\n    image = tf.reshape(image, [IMAGE_SIZE[0], IMAGE_SIZE[1], 3])\n    return image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:31.123382Z","iopub.execute_input":"2025-10-12T08:49:31.124147Z","iopub.status.idle":"2025-10-12T08:49:31.129773Z","shell.execute_reply.started":"2025-10-12T08:49:31.124109Z","shell.execute_reply":"2025-10-12T08:49:31.128546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_images[0].shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:31.131116Z","iopub.execute_input":"2025-10-12T08:49:31.131580Z","iopub.status.idle":"2025-10-12T08:49:31.144494Z","shell.execute_reply.started":"2025-10-12T08:49:31.131544Z","shell.execute_reply":"2025-10-12T08:49:31.143446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_files","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:31.146002Z","iopub.execute_input":"2025-10-12T08:49:31.146379Z","iopub.status.idle":"2025-10-12T08:49:31.155623Z","shell.execute_reply.started":"2025-10-12T08:49:31.146342Z","shell.execute_reply":"2025-10-12T08:49:31.154495Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_picked = training_files[0]\nsample_picked","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:31.156756Z","iopub.execute_input":"2025-10-12T08:49:31.157064Z","iopub.status.idle":"2025-10-12T08:49:31.167571Z","shell.execute_reply.started":"2025-10-12T08:49:31.157041Z","shell.execute_reply":"2025-10-12T08:49:31.165712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"file = tf.data.TFRecordDataset(sample_picked)\nfile","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:31.169413Z","iopub.execute_input":"2025-10-12T08:49:31.169977Z","iopub.status.idle":"2025-10-12T08:49:34.672031Z","shell.execute_reply.started":"2025-10-12T08:49:31.169944Z","shell.execute_reply":"2025-10-12T08:49:34.671113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_description = {\"image\" : tf.io.FixedLenFeature([], tf.string), \n                      \"target\" : tf.io.FixedLenFeature([], tf.int64)}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:34.673206Z","iopub.execute_input":"2025-10-12T08:49:34.673466Z","iopub.status.idle":"2025-10-12T08:49:34.677887Z","shell.execute_reply.started":"2025-10-12T08:49:34.673437Z","shell.execute_reply":"2025-10-12T08:49:34.676990Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def parse_function(example) : \n    # The example supplied is parsed based on the feature_description above.\n    return tf.io.parse_single_example(example, feature_description)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:34.679201Z","iopub.execute_input":"2025-10-12T08:49:34.679562Z","iopub.status.idle":"2025-10-12T08:49:34.687379Z","shell.execute_reply.started":"2025-10-12T08:49:34.679527Z","shell.execute_reply":"2025-10-12T08:49:34.686549Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"parsed_dataset = file.map(parse_function)\nparsed_dataset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:34.688340Z","iopub.execute_input":"2025-10-12T08:49:34.688681Z","iopub.status.idle":"2025-10-12T08:49:34.726579Z","shell.execute_reply.started":"2025-10-12T08:49:34.688646Z","shell.execute_reply":"2025-10-12T08:49:34.725790Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def read_tfrecord(example, labeled) : \n    if labeled == True : \n        tfrecord_format = {\"image\" : tf.io.FixedLenFeature([], tf.string),\n                           \"target\" : tf.io.FixedLenFeature([], tf.int64)}\n    else:\n        tfrecord_format = {\"image\" : tf.io.FixedLenFeature([], tf.string),\n                          \"image_name\" : tf.io.FixedLenFeature([], tf.string)}\n    \n    example = tf.io.parse_single_example(example, tfrecord_format)\n    image = decode_image(example[\"image\"])\n    if labeled == True : \n        label = tf.cast(example[\"target\"], tf.int32)\n        return image, label\n    else:\n        image_name = example[\"image_name\"]\n        return image, image_name     ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:34.727490Z","iopub.execute_input":"2025-10-12T08:49:34.727756Z","iopub.status.idle":"2025-10-12T08:49:34.733650Z","shell.execute_reply.started":"2025-10-12T08:49:34.727710Z","shell.execute_reply":"2025-10-12T08:49:34.732781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_dataset(filenames, labeled, ordered):\n    ignore_order = tf.data.Options()\n    if ordered == False: # dataset is unordered, so we ignore the order to load data quickly.\n        ignore_order.experimental_deterministic = False # This disables the order and enhances the speed\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTOTUNE) \n    dataset = dataset.with_options(ignore_order) \n    dataset = dataset.map(partial(read_tfrecord, labeled=labeled), num_parallel_calls=AUTOTUNE)\n    return dataset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:34.734676Z","iopub.execute_input":"2025-10-12T08:49:34.734988Z","iopub.status.idle":"2025-10-12T08:49:34.742052Z","shell.execute_reply.started":"2025-10-12T08:49:34.734965Z","shell.execute_reply":"2025-10-12T08:49:34.741235Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def image_augmentation(image, label) :     \n    image = tf.image.resize(image, SHAPE)\n    image = tf.image.random_flip_left_right(image)\n    return image, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:34.743034Z","iopub.execute_input":"2025-10-12T08:49:34.743381Z","iopub.status.idle":"2025-10-12T08:49:34.752350Z","shell.execute_reply.started":"2025-10-12T08:49:34.743356Z","shell.execute_reply":"2025-10-12T08:49:34.751567Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load The Datasets : ","metadata":{}},{"cell_type":"code","source":"def get_training_dataset() : \n    dataset = load_dataset(training_files, labeled = True, ordered = False)\n    dataset = dataset.map(image_augmentation, num_parallel_calls=AUTOTUNE)\n    dataset = dataset.repeat()\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTOTUNE) \n    return dataset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:34.753341Z","iopub.execute_input":"2025-10-12T08:49:34.753680Z","iopub.status.idle":"2025-10-12T08:49:34.760638Z","shell.execute_reply.started":"2025-10-12T08:49:34.753645Z","shell.execute_reply":"2025-10-12T08:49:34.759764Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_validation_dataset() : \n    dataset = load_dataset(validation_files, labeled = True, ordered = False)\n    dataset = dataset.map(image_augmentation, num_parallel_calls=AUTOTUNE)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.cache()\n    dataset = dataset.prefetch(AUTOTUNE) \n    return dataset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:34.761912Z","iopub.execute_input":"2025-10-12T08:49:34.762421Z","iopub.status.idle":"2025-10-12T08:49:34.768447Z","shell.execute_reply.started":"2025-10-12T08:49:34.762378Z","shell.execute_reply":"2025-10-12T08:49:34.767525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_test_dataset() : \n    dataset = load_dataset(testing_files, labeled = False, ordered = True)\n    dataset = dataset.map(image_augmentation, num_parallel_calls=AUTOTUNE)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.cache()\n    dataset = dataset.prefetch(AUTOTUNE) \n    return dataset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:34.769589Z","iopub.execute_input":"2025-10-12T08:49:34.769990Z","iopub.status.idle":"2025-10-12T08:49:34.775905Z","shell.execute_reply.started":"2025-10-12T08:49:34.769966Z","shell.execute_reply":"2025-10-12T08:49:34.775162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_dataset = get_training_dataset()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:34.776848Z","iopub.execute_input":"2025-10-12T08:49:34.777122Z","iopub.status.idle":"2025-10-12T08:49:34.966990Z","shell.execute_reply.started":"2025-10-12T08:49:34.777099Z","shell.execute_reply":"2025-10-12T08:49:34.966093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"validation_dataset = get_validation_dataset()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:34.967968Z","iopub.execute_input":"2025-10-12T08:49:34.968219Z","iopub.status.idle":"2025-10-12T08:49:35.004888Z","shell.execute_reply.started":"2025-10-12T08:49:34.968196Z","shell.execute_reply":"2025-10-12T08:49:35.004041Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def count_data_items(filenames):\n    n = [int(re.compile(r\"-([0-9]*)\\.\").search(filename).group(1)) for filename in filenames]\n    return np.sum(n)\n\nnum_training_images = count_data_items(training_files)\nnum_validation_images = count_data_items(validation_files)\nnum_testing_images = count_data_items(testing_files)\n\nSTEPS_PER_EPOCH_TRAIN = num_training_images // BATCH_SIZE\nSTEPS_PER_EPOCH_VAL = num_validation_images // BATCH_SIZE\n\nprint(\"Number of Training Images = \", num_training_images)\nprint(\"Number of Validation Images = \", num_validation_images)\nprint(\"Number of Testing Images = \", num_testing_images)\nprint(\"\\n\")\nprint(\"Numer of steps per epoch in Train = \", STEPS_PER_EPOCH_TRAIN)\nprint(\"Numer of steps per epoch in Validation = \", STEPS_PER_EPOCH_VAL)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:35.005983Z","iopub.execute_input":"2025-10-12T08:49:35.006233Z","iopub.status.idle":"2025-10-12T08:49:35.013217Z","shell.execute_reply.started":"2025-10-12T08:49:35.006211Z","shell.execute_reply":"2025-10-12T08:49:35.012347Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_batch, label_batch = next(iter(training_dataset))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:35.014206Z","iopub.execute_input":"2025-10-12T08:49:35.014444Z","iopub.status.idle":"2025-10-12T08:49:37.050684Z","shell.execute_reply.started":"2025-10-12T08:49:35.014422Z","shell.execute_reply":"2025-10-12T08:49:37.049932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def show_batch(image_batch, label_batch) :\n    plt.figure(figsize = (20, 20))\n    for n in range(8) : \n        ax = plt.subplot(2,4,n+1)\n        plt.imshow(image_batch[n])\n        if label_batch[n] == 0 : \n            plt.title(\"BENIGN\")\n        else:\n            plt.title(\"MALIGNANT\")\n    plt.grid(False)\n    plt.tight_layout()       ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:37.051743Z","iopub.execute_input":"2025-10-12T08:49:37.052032Z","iopub.status.idle":"2025-10-12T08:49:37.057673Z","shell.execute_reply.started":"2025-10-12T08:49:37.052006Z","shell.execute_reply":"2025-10-12T08:49:37.056775Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"show_batch(image_batch.numpy(), label_batch.numpy())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:37.058694Z","iopub.execute_input":"2025-10-12T08:49:37.058969Z","iopub.status.idle":"2025-10-12T08:49:38.887445Z","shell.execute_reply.started":"2025-10-12T08:49:37.058934Z","shell.execute_reply":"2025-10-12T08:49:38.886179Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Let's free up some memory","metadata":{}},{"cell_type":"code","source":"del image_batch\ndel label_batch\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:38.889124Z","iopub.execute_input":"2025-10-12T08:49:38.889398Z","iopub.status.idle":"2025-10-12T08:49:39.073685Z","shell.execute_reply.started":"2025-10-12T08:49:38.889373Z","shell.execute_reply":"2025-10-12T08:49:39.072778Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Construction : ","metadata":{}},{"cell_type":"code","source":"malignant = len(train[train[\"target\"] == 1])\nbenign = len(train[train[\"target\"] == 0 ])\ntotal = len(train) \n\nprint(\"Malignant Cases in Train Data = \", malignant)\nprint(\"Benign Cases In Train Dataset = \",benign)\nprint(\"Total Cases In Train Dataset = \",total)\nprint(\"Ratio of Malignant to Benign = \",malignant/benign)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:39.074889Z","iopub.execute_input":"2025-10-12T08:49:39.075213Z","iopub.status.idle":"2025-10-12T08:49:39.087513Z","shell.execute_reply.started":"2025-10-12T08:49:39.075156Z","shell.execute_reply":"2025-10-12T08:49:39.086684Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"weight_malignant = (total/malignant)/2.0\nweight_benign = (total/benign)/2.0\n\nclass_weight = {0 : weight_benign , 1 : weight_malignant}\n\nprint(\"Weight for benign cases = \", class_weight[0])\nprint(\"Weight for malignant cases = \", class_weight[1])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:39.088617Z","iopub.execute_input":"2025-10-12T08:49:39.088976Z","iopub.status.idle":"2025-10-12T08:49:39.094671Z","shell.execute_reply.started":"2025-10-12T08:49:39.088944Z","shell.execute_reply":"2025-10-12T08:49:39.093691Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"callback_early_stopping = tf.keras.callbacks.EarlyStopping(patience = 15, verbose = 0, restore_best_weights = True)\n\ncallbacks_lr_reduce = tf.keras.callbacks.ReduceLROnPlateau(monitor = \"val_auc\", factor = 0.1, patience = 10, \n                                                          verbose = 0, min_lr = 1e-6)\n\ncallback_checkpoint = tf.keras.callbacks.ModelCheckpoint(\"melanoma_weights.h5\",\n                                                         save_weights_only=True, monitor='val_auc',\n                                                         mode='max', save_best_only = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:39.095952Z","iopub.execute_input":"2025-10-12T08:49:39.096322Z","iopub.status.idle":"2025-10-12T08:49:39.862996Z","shell.execute_reply.started":"2025-10-12T08:49:39.096292Z","shell.execute_reply":"2025-10-12T08:49:39.862190Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model Design : MobileNetV2\n\nA supercool resource : **https://machinethink.net/blog/mobilenet-v2/**","metadata":{},"attachments":{}},{"cell_type":"markdown","source":"## Bias Initialization : \n\nSince the dataset is heavily imbalanced, we may want to assign different weights to different classes. Setting an initial bias is important in such cases.","metadata":{}},{"cell_type":"code","source":"with strategy.scope() : \n    bias = np.log(malignant/benign)\n    bias = tf.keras.initializers.Constant(bias)\n    base_model = tf.keras.applications.MobileNetV2(input_shape = (SHAPE[0], SHAPE[1], 3), include_top = False,\n                                               weights = \"imagenet\")\n    base_model.trainable = False\n    model = tf.keras.Sequential([base_model,\n                                 tf.keras.layers.GlobalAveragePooling2D(),\n                                 tf.keras.layers.Dense(20, activation = \"relu\"),\n                                 tf.keras.layers.Dropout(0.4),\n                                 tf.keras.layers.Dense(10, activation = \"relu\"),\n                                 tf.keras.layers.Dropout(0.3),\n                                 tf.keras.layers.Dense(1, activation = \"sigmoid\", bias_initializer = bias)                                     \n                                ])\n    model.compile(optimizer = tf.keras.optimizers.Adam(lr = 1e-2), loss = \"binary_crossentropy\", metrics = [tf.keras.metrics.AUC(name = 'auc')])\n    model.summary()\n    \n    EPOCHS = 500\n    history = model.fit(training_dataset, epochs = EPOCHS, steps_per_epoch = STEPS_PER_EPOCH_TRAIN,\n                       validation_data = validation_dataset, validation_steps = STEPS_PER_EPOCH_VAL,\n                       callbacks = [callback_early_stopping, callbacks_lr_reduce, callback_checkpoint],\n                       class_weight = class_weight)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T08:49:39.864009Z","iopub.execute_input":"2025-10-12T08:49:39.864274Z","iopub.status.idle":"2025-10-12T10:05:31.653764Z","shell.execute_reply.started":"2025-10-12T08:49:39.864250Z","shell.execute_reply":"2025-10-12T10:05:31.652785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"n_epochs_it_ran_for = len(history.history['loss'])\nn_epochs_it_ran_for","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T10:05:31.655132Z","iopub.execute_input":"2025-10-12T10:05:31.655508Z","iopub.status.idle":"2025-10-12T10:05:31.662772Z","shell.execute_reply.started":"2025-10-12T10:05:31.655467Z","shell.execute_reply":"2025-10-12T10:05:31.661583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = np.arange(0,n_epochs_it_ran_for,1)\nplt.figure(1, figsize = (20, 12))\nplt.subplot(1,2,1)\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.plot(X, history.history[\"loss\"], label = \"Training Loss\")\nplt.plot(X, history.history[\"val_loss\"], label = \"Validation Loss\")\nplt.grid(True)\nplt.legend()\n\nplt.subplot(1,2,2)\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Accuracy\")\nplt.plot(X, history.history[\"auc\"], label = \"Training Accuracy\")\nplt.plot(X, history.history[\"val_auc\"], label = \"Validation Accuracy\")\nplt.grid(True)\nplt.legend()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T10:05:31.663866Z","iopub.execute_input":"2025-10-12T10:05:31.664266Z","iopub.status.idle":"2025-10-12T10:06:06.082618Z","shell.execute_reply.started":"2025-10-12T10:05:31.664223Z","shell.execute_reply":"2025-10-12T10:06:06.081693Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Due to callbacks, best weights are automatically restored!","metadata":{}},{"cell_type":"code","source":"testing_dataset = get_test_dataset()\ntesting_dataset_images = testing_dataset.map(lambda image, image_name : image)\ntesting_image_names = testing_dataset.map(lambda image, image_name : image_name)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T10:06:06.083710Z","iopub.execute_input":"2025-10-12T10:06:06.084007Z","iopub.status.idle":"2025-10-12T10:06:06.147374Z","shell.execute_reply.started":"2025-10-12T10:06:06.083983Z","shell.execute_reply":"2025-10-12T10:06:06.146676Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"resulting_probabilities = model.predict(testing_dataset_images, verbose = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T10:06:06.148337Z","iopub.execute_input":"2025-10-12T10:06:06.148577Z","iopub.status.idle":"2025-10-12T10:07:58.085048Z","shell.execute_reply.started":"2025-10-12T10:06:06.148555Z","shell.execute_reply":"2025-10-12T10:07:58.084173Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(resulting_probabilities)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T10:07:58.086100Z","iopub.execute_input":"2025-10-12T10:07:58.086401Z","iopub.status.idle":"2025-10-12T10:07:58.092510Z","shell.execute_reply.started":"2025-10-12T10:07:58.086375Z","shell.execute_reply":"2025-10-12T10:07:58.091556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_submission_file = pd.read_csv(\"../input/siim-isic-melanoma-classification/sample_submission.csv\")\nsample_submission_file.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T10:07:58.093538Z","iopub.execute_input":"2025-10-12T10:07:58.093855Z","iopub.status.idle":"2025-10-12T10:07:58.133544Z","shell.execute_reply.started":"2025-10-12T10:07:58.093819Z","shell.execute_reply":"2025-10-12T10:07:58.132782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del sample_submission_file[\"target\"]\nsample_submission_file.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T10:07:58.134504Z","iopub.execute_input":"2025-10-12T10:07:58.134773Z","iopub.status.idle":"2025-10-12T10:07:58.143536Z","shell.execute_reply.started":"2025-10-12T10:07:58.134746Z","shell.execute_reply":"2025-10-12T10:07:58.142581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"testing_image_names","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T10:07:58.144848Z","iopub.execute_input":"2025-10-12T10:07:58.145220Z","iopub.status.idle":"2025-10-12T10:07:58.151623Z","shell.execute_reply.started":"2025-10-12T10:07:58.145184Z","shell.execute_reply":"2025-10-12T10:07:58.150739Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"testing_image_names = np.concatenate([x for x in testing_image_names], axis=0)\ntesting_image_names = np.array(testing_image_names)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T10:07:58.152887Z","iopub.execute_input":"2025-10-12T10:07:58.153463Z","iopub.status.idle":"2025-10-12T10:07:58.252355Z","shell.execute_reply.started":"2025-10-12T10:07:58.153420Z","shell.execute_reply":"2025-10-12T10:07:58.251674Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"decoded_test_names = []\nfor names in testing_image_names : \n    names = names.decode('utf-8')\n    decoded_test_names.append(names)\ndecoded_test_names = np.array(decoded_test_names)\ndel testing_image_names","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T10:07:58.253342Z","iopub.execute_input":"2025-10-12T10:07:58.253673Z","iopub.status.idle":"2025-10-12T10:07:58.263593Z","shell.execute_reply.started":"2025-10-12T10:07:58.253638Z","shell.execute_reply":"2025-10-12T10:07:58.262876Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(decoded_test_names), type(decoded_test_names), decoded_test_names.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T10:07:58.264714Z","iopub.execute_input":"2025-10-12T10:07:58.265177Z","iopub.status.idle":"2025-10-12T10:07:58.271580Z","shell.execute_reply.started":"2025-10-12T10:07:58.265141Z","shell.execute_reply":"2025-10-12T10:07:58.270697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"decoded_test_names","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T10:07:58.272626Z","iopub.execute_input":"2025-10-12T10:07:58.272878Z","iopub.status.idle":"2025-10-12T10:07:58.278890Z","shell.execute_reply.started":"2025-10-12T10:07:58.272856Z","shell.execute_reply":"2025-10-12T10:07:58.277926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"testing_image_names = pd.DataFrame(decoded_test_names, columns=[\"image_name\"])\ntesting_image_names.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T10:07:58.279823Z","iopub.execute_input":"2025-10-12T10:07:58.280108Z","iopub.status.idle":"2025-10-12T10:07:58.290955Z","shell.execute_reply.started":"2025-10-12T10:07:58.280086Z","shell.execute_reply":"2025-10-12T10:07:58.290046Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred_dataframe = pd.DataFrame({\"image_name\" : decoded_test_names, \n                               \"target\" : np.concatenate(resulting_probabilities)})\npred_dataframe","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T10:07:58.291877Z","iopub.execute_input":"2025-10-12T10:07:58.292114Z","iopub.status.idle":"2025-10-12T10:07:58.315408Z","shell.execute_reply.started":"2025-10-12T10:07:58.292093Z","shell.execute_reply":"2025-10-12T10:07:58.314620Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_submission_file = sample_submission_file.merge(pred_dataframe, on = \"image_name\")\nsample_submission_file.to_csv(\"submission.csv\", index = False)\nsample_submission_file.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T10:07:58.316303Z","iopub.execute_input":"2025-10-12T10:07:58.316525Z","iopub.status.idle":"2025-10-12T10:07:58.363634Z","shell.execute_reply.started":"2025-10-12T10:07:58.316506Z","shell.execute_reply":"2025-10-12T10:07:58.362836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.save(\"melanoma_model.h5\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T10:41:09.202622Z","iopub.execute_input":"2025-10-12T10:41:09.203010Z","iopub.status.idle":"2025-10-12T10:41:09.415353Z","shell.execute_reply.started":"2025-10-12T10:41:09.202979Z","shell.execute_reply":"2025-10-12T10:41:09.414448Z"}},"outputs":[],"execution_count":null}]}