{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"}],"dockerImageVersionId":30840,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport os\nimport cv2\nimport random\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report,roc_curve\n\nfrom tensorflow.keras.utils import Sequence\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import VGG16,ResNet50\nfrom tensorflow.keras import layers\nfrom tensorflow.keras import Input,Model\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import backend as K","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T12:46:29.116819Z","iopub.execute_input":"2025-02-04T12:46:29.117134Z","iopub.status.idle":"2025-02-04T12:46:32.975740Z","shell.execute_reply.started":"2025-02-04T12:46:29.117110Z","shell.execute_reply":"2025-02-04T12:46:32.974993Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 1. Data Preprocessing","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')\ndata.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T12:46:32.976790Z","iopub.execute_input":"2025-02-04T12:46:32.977328Z","iopub.status.idle":"2025-02-04T12:46:33.047844Z","shell.execute_reply.started":"2025-02-04T12:46:32.977303Z","shell.execute_reply":"2025-02-04T12:46:33.046973Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### I've added a new column that contains the file paths to the images","metadata":{}},{"cell_type":"code","source":"image_path = []\n\npattern = '../input/siim-isic-melanoma-classification/jpeg/train'\nfor i in data['image_name'].values:\n    path = os.path.join(pattern,i)\n    path += '.jpg'\n    image_path.append(path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T12:46:33.049268Z","iopub.execute_input":"2025-02-04T12:46:33.049607Z","iopub.status.idle":"2025-02-04T12:46:33.098497Z","shell.execute_reply.started":"2025-02-04T12:46:33.049579Z","shell.execute_reply":"2025-02-04T12:46:33.097725Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data['image_path'] = image_path","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T12:46:33.099861Z","iopub.execute_input":"2025-02-04T12:46:33.100201Z","iopub.status.idle":"2025-02-04T12:46:33.108867Z","shell.execute_reply.started":"2025-02-04T12:46:33.100175Z","shell.execute_reply":"2025-02-04T12:46:33.108002Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Let's take a look at the initial dataset","metadata":{}},{"cell_type":"code","source":"sns.countplot(data,x='target')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T12:46:33.109712Z","iopub.execute_input":"2025-02-04T12:46:33.110050Z","iopub.status.idle":"2025-02-04T12:46:33.261218Z","shell.execute_reply.started":"2025-02-04T12:46:33.110016Z","shell.execute_reply":"2025-02-04T12:46:33.260098Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data['target'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T12:46:33.262400Z","iopub.execute_input":"2025-02-04T12:46:33.262795Z","iopub.status.idle":"2025-02-04T12:46:33.269911Z","shell.execute_reply.started":"2025-02-04T12:46:33.262760Z","shell.execute_reply":"2025-02-04T12:46:33.269000Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### This dataset is highly unbalanced, so we applied downsampling to the 'healthy' label samples and used weight initialization techniques to address the class imbalance.","metadata":{}},{"cell_type":"code","source":"data_class_0 = data[data.target==0].sample(3000,random_state=42)\ndata_class_1 = data[data.target==1]\nnew_data = pd.concat([data_class_0,data_class_1])\nsns.countplot(new_data,x='target')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T12:46:33.270898Z","iopub.execute_input":"2025-02-04T12:46:33.271257Z","iopub.status.idle":"2025-02-04T12:46:33.413706Z","shell.execute_reply.started":"2025-02-04T12:46:33.271222Z","shell.execute_reply":"2025-02-04T12:46:33.412732Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### After downsampling, the dataset became more balanced, helping to mitigate the risk of overfitting.","metadata":{}},{"cell_type":"code","source":"new_data['target'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T12:46:33.416110Z","iopub.execute_input":"2025-02-04T12:46:33.416346Z","iopub.status.idle":"2025-02-04T12:46:33.422810Z","shell.execute_reply.started":"2025-02-04T12:46:33.416326Z","shell.execute_reply":"2025-02-04T12:46:33.422052Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 2. Creating the Dataset class","metadata":{}},{"cell_type":"code","source":"class Data(Sequence):\n    def __init__(self, image_path, target, batch_size, target_size=(224, 224), aug=None, shuffle=True, seed=42,**kwargs):\n        super().__init__()\n        self.image_path = np.array(image_path)\n        self.target = np.array(target)\n        self.batch_size = batch_size\n        self.target_size = target_size\n        self.aug = aug\n        self.shuffle = shuffle\n        self.seed = seed\n        np.random.seed(self.seed) \n        self.on_epoch_end()\n    \n    def __len__(self):\n        return int(np.ceil(len(self.image_path) / self.batch_size))  \n\n    def __getitem__(self, item):\n        batch_indices = self.indices[item * self.batch_size: (item + 1) * self.batch_size]\n        image_path_batch = [self.image_path[i] for i in batch_indices]\n        label_batch = [self.target[i] for i in batch_indices]\n        images = [self.load_data(i) for i in image_path_batch]\n\n        images = np.array(images)\n        label_batch = np.array(label_batch)\n        \n        if self.aug:\n            auged = self.aug.flow(images, label_batch, shuffle=False)  \n            images, label_batch = next(auged)\n\n        return images, label_batch\n\n    def on_epoch_end(self):\n        self.indices = np.arange(len(self.image_path))\n        if self.shuffle:\n            np.random.shuffle(self.indices)  \n\n    def load_data(self, image_path):\n        img = image.load_img(image_path,target_size=self.target_size) \n        img = image.img_to_array(img)\n        img /= 255.0\n        return img","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T12:46:33.424555Z","iopub.execute_input":"2025-02-04T12:46:33.424815Z","iopub.status.idle":"2025-02-04T12:46:33.440537Z","shell.execute_reply.started":"2025-02-04T12:46:33.424793Z","shell.execute_reply":"2025-02-04T12:46:33.439704Z"},"_kg_hide-output":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Using the aug to apply image augmentation to the loaded data, which helps increase dataset diversity and improve model generalization.","metadata":{}},{"cell_type":"code","source":"def load_and_preprocess_train(image_path, label, target_size=(224, 224)):\n    img = tf.io.read_file(image_path)\n    img = tf.image.decode_jpeg(img, channels=3)\n    img = tf.image.resize(img, target_size)\n    img = tf.image.random_flip_left_right(img) \n    img = tf.image.random_flip_up_down(img)    \n    img = tf.image.random_crop(img, size=[target_size[0], target_size[1], 3])\n    img = img - [123.68, 116.78, 103.94] \n    img = img / [58.40, 57.12, 57.37] \n    \n    return img, label\n\ndef load_and_preprocess_valid(image_path, label, target_size=(224, 224)):\n    img = tf.io.read_file(image_path)\n    img = tf.image.decode_jpeg(img, channels=3)\n    img = tf.image.resize(img, target_size)\n    img = img - [123.68, 116.78, 103.94]  \n    img = img / [58.40, 57.12, 57.37] \n    \n    return img, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T12:46:33.441431Z","iopub.execute_input":"2025-02-04T12:46:33.441864Z","iopub.status.idle":"2025-02-04T12:46:33.461188Z","shell.execute_reply.started":"2025-02-04T12:46:33.441822Z","shell.execute_reply":"2025-02-04T12:46:33.460304Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"datagen = ImageDataGenerator(\n    rotation_range=30,  \n    width_shift_range=0.2,  \n    height_shift_range=0.2, \n    shear_range=0.2,     \n    zoom_range=0.2,  \n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\ndataset = Data(new_data['image_path'].values,new_data['target'].values,batch_size=32,target_size=(224,224),aug=datagen)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T12:46:33.462084Z","iopub.execute_input":"2025-02-04T12:46:33.462435Z","iopub.status.idle":"2025-02-04T12:46:33.480865Z","shell.execute_reply.started":"2025-02-04T12:46:33.462405Z","shell.execute_reply":"2025-02-04T12:46:33.480007Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for images,labels in dataset:\n    fig, ax = plt.subplots(4,8,figsize=(12,6))\n    ax = ax.flatten()\n\n    for value,ax_i in enumerate(ax):\n        ax_i.imshow(images[value])\n        ax_i.set_title(labels[value])\n        ax_i.axis('off')\n    break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T12:46:33.481709Z","iopub.execute_input":"2025-02-04T12:46:33.481948Z","iopub.status.idle":"2025-02-04T12:46:38.895140Z","shell.execute_reply.started":"2025-02-04T12:46:33.481928Z","shell.execute_reply":"2025-02-04T12:46:38.894142Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 3. Building the Model","metadata":{}},{"cell_type":"markdown","source":"#### Using the VGG16 architecture with pre-trained weights, only modifying the fully connected (FC) layers.","metadata":{}},{"cell_type":"code","source":"# def binary_focal_loss(alpha=0.25, gamma=2):\n#     def loss(y_true, y_pred):\n#         bce = K.binary_crossentropy(y_true, y_pred)\n#         p_t = y_true * y_pred + (1 - y_true) * (1 - y_pred)\n#         alpha_factor = y_true * alpha + (1 - y_true) * (1 - alpha)\n#         modulating_factor = K.pow(1 - p_t, gamma)\n#         focal_loss = alpha_factor * modulating_factor * bce\n#         return K.mean(focal_loss)\n#     return loss\n\ndef create_model(lr=0.0001):\n    base_model = VGG16(input_shape=(224,224,3),include_top = False)\n\n    for layer in base_model.layers:\n        layer.trainable = False\n    \n    input = base_model.layers[-1].output\n    x = layers.GlobalAveragePooling2D()(input)\n    x = layers.Dense(512, activation = 'relu')(x)\n    output = layers.Dense(1, activation = 'sigmoid')(x)\n\n    model = Model(base_model.input,output)\n    print(model.summary())\n    model.compile(\n        #loss = 'binary_crossentropy',\n        loss = tf.keras.losses.BinaryFocalCrossentropy(alpha=0.25,gamma=2.0),\n        #binary_focal_loss(alpha=0.2,gamma=2),\n        #metrics=[keras.metrics.Recall()],\n        metrics = ['acc'],\n        optimizer = keras.optimizers.Adam(learning_rate = lr),\n    )\n    return model\n\nmodel = create_model()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T12:55:26.948139Z","iopub.execute_input":"2025-02-04T12:55:26.948460Z","iopub.status.idle":"2025-02-04T12:55:27.251613Z","shell.execute_reply.started":"2025-02-04T12:55:26.948436Z","shell.execute_reply":"2025-02-04T12:55:27.250710Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 4. Train","metadata":{}},{"cell_type":"code","source":"x_train,x_valid,y_train,y_valid = train_test_split(new_data['image_path'],new_data['target'].values,\n                                                  test_size=0.1,\n                                                  random_state=42,\n                                                  stratify=new_data['target'].values)\n\nprint(x_train.shape,y_train.shape)\nprint(x_valid.shape,y_valid.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T12:46:40.450283Z","iopub.execute_input":"2025-02-04T12:46:40.450520Z","iopub.status.idle":"2025-02-04T12:46:40.459322Z","shell.execute_reply.started":"2025-02-04T12:46:40.450499Z","shell.execute_reply":"2025-02-04T12:46:40.458520Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#train_generator = Data(x_train, y_train, batch_size=32, aug=datagen, shuffle=True)\n#valid_generator = Data(x_valid, y_valid, batch_size=32, aug=None, shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T12:46:40.460339Z","iopub.execute_input":"2025-02-04T12:46:40.460735Z","iopub.status.idle":"2025-02-04T12:46:40.474141Z","shell.execute_reply.started":"2025-02-04T12:46:40.460700Z","shell.execute_reply":"2025-02-04T12:46:40.473388Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"batch_size = 32\n\ntrain_dataset = tf.data.Dataset.from_tensor_slices((x_train, y_train))\ntrain_dataset = train_dataset.map(lambda x, y: load_and_preprocess_train(x, y, target_size=(224, 224)), num_parallel_calls=tf.data.AUTOTUNE)\ntrain_dataset = train_dataset.batch(batch_size).prefetch(tf.data.AUTOTUNE)\n\nvalid_dataset = tf.data.Dataset.from_tensor_slices((x_valid, y_valid))\nvalid_dataset = valid_dataset.map(lambda x, y: load_and_preprocess_valid(x, y, target_size=(224, 224)), num_parallel_calls=tf.data.AUTOTUNE)\nvalid_dataset = valid_dataset.batch(batch_size).prefetch(tf.data.AUTOTUNE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T12:46:40.474964Z","iopub.execute_input":"2025-02-04T12:46:40.475270Z","iopub.status.idle":"2025-02-04T12:46:40.693451Z","shell.execute_reply.started":"2025-02-04T12:46:40.475241Z","shell.execute_reply":"2025-02-04T12:46:40.692445Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Initialized the weights according to the label distribution to further address the class imbalance (when not using focal loss)","metadata":{}},{"cell_type":"code","source":"class_0_weight = len(y_train) / (2 * np.bincount(y_train)[0])\nclass_1_weight = len(y_train) / (2 * np.bincount(y_train)[1])\nclass_weight = {0: class_0_weight,1:class_1_weight}\nprint(class_weight)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T12:50:30.213424Z","iopub.execute_input":"2025-02-04T12:50:30.213795Z","iopub.status.idle":"2025-02-04T12:50:30.219008Z","shell.execute_reply.started":"2025-02-04T12:50:30.213765Z","shell.execute_reply":"2025-02-04T12:50:30.217943Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Applying early stopping to prevent overfitting and adjusting the learning rate to accelerate convergence","metadata":{}},{"cell_type":"code","source":"epochs = 100\nfactor = 0.2\n\nearly_stopping = keras.callbacks.EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True)\nreduce_lr = keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor= factor, patience=3, min_lr=1e-7)\n\nH = model.fit(\n    train_dataset,#class_weight = class_weight,\n    validation_data = valid_dataset,\n    epochs = epochs,\n    callbacks = [early_stopping , reduce_lr]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T12:55:31.618463Z","iopub.execute_input":"2025-02-04T12:55:31.618863Z","iopub.status.idle":"2025-02-04T13:11:48.297116Z","shell.execute_reply.started":"2025-02-04T12:55:31.618828Z","shell.execute_reply":"2025-02-04T13:11:48.291666Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig , ax = plt.subplots(1,2, figsize=(10,6))\n\ntrain_acc = H.history['acc']\nval_acc = H.history['val_acc']\ntrain_loss = H.history['loss']\nval_loss = H.history['val_loss']\n\nnum_epoch = len(train_acc)\nax[0].plot(range(1,num_epoch+1) , train_acc , label = 'Train')\nax[0].plot(range(1,num_epoch+1) , val_acc, label = 'Val')\nax[0].set_xlabel('Epochs')\nax[0].set_ylabel('Accuracy')\nax[0].legend()\n\nax[1].plot(range(1,num_epoch+1) , train_loss , label = 'Train')\nax[1].plot(range(1,num_epoch+1) , val_loss, label = 'Val')\nax[1].set_xlabel('Epochs')\nax[1].set_ylabel('Loss')\nax[1].legend()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T13:11:52.785325Z","iopub.execute_input":"2025-02-04T13:11:52.785668Z","iopub.status.idle":"2025-02-04T13:11:53.058098Z","shell.execute_reply.started":"2025-02-04T13:11:52.785609Z","shell.execute_reply":"2025-02-04T13:11:53.056941Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = model.predict(valid_dataset)\ny_pred = np.where(y_pred>0.2,1,0)\n\nreport = classification_report(y_valid,y_pred)\nprint(report)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T13:12:11.489202Z","iopub.execute_input":"2025-02-04T13:12:11.489506Z","iopub.status.idle":"2025-02-04T13:12:21.905496Z","shell.execute_reply.started":"2025-02-04T13:12:11.489483Z","shell.execute_reply":"2025-02-04T13:12:21.904723Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Submit","metadata":{}},{"cell_type":"code","source":"data_test = pd.read_csv('../input/siim-isic-melanoma-classification/test.csv')\ndata_test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T13:17:34.756621Z","iopub.execute_input":"2025-02-04T13:17:34.757001Z","iopub.status.idle":"2025-02-04T13:17:34.790084Z","shell.execute_reply.started":"2025-02-04T13:17:34.756973Z","shell.execute_reply":"2025-02-04T13:17:34.789274Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_path = []\n\npattern = '../input/siim-isic-melanoma-classification/jpeg/test'\nfor i in data_test['image_name'].values:\n    path = os.path.join(pattern,i)\n    path += '.jpg'\n    image_path.append(path)\n\ndata_test['image_path'] = image_path","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T13:17:37.527790Z","iopub.execute_input":"2025-02-04T13:17:37.528107Z","iopub.status.idle":"2025-02-04T13:17:37.548376Z","shell.execute_reply.started":"2025-02-04T13:17:37.528085Z","shell.execute_reply":"2025-02-04T13:17:37.547425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"valid_dataset = tf.data.Dataset.from_tensor_slices((data_test['image_path'].values, None))\nvalid_dataset = valid_dataset.map(lambda x, _: load_and_preprocess_valid(x, _, target_size=(224, 224)), num_parallel_calls=tf.data.AUTOTUNE)\nvalid_dataset = valid_dataset.batch(batch_size).prefetch(tf.data.AUTOTUNE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T13:17:39.762495Z","iopub.execute_input":"2025-02-04T13:17:39.762882Z","iopub.status.idle":"2025-02-04T13:17:39.800133Z","shell.execute_reply.started":"2025-02-04T13:17:39.762852Z","shell.execute_reply":"2025-02-04T13:17:39.799397Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"prediction = model.predict(valid_dataset)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T13:17:42.026593Z","iopub.execute_input":"2025-02-04T13:17:42.026972Z","iopub.status.idle":"2025-02-04T13:22:50.041414Z","shell.execute_reply.started":"2025-02-04T13:17:42.026943Z","shell.execute_reply":"2025-02-04T13:22:50.040676Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"prediction_ = [i[0] for i in prediction]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T13:22:50.042588Z","iopub.execute_input":"2025-02-04T13:22:50.042893Z","iopub.status.idle":"2025-02-04T13:22:50.050008Z","shell.execute_reply.started":"2025-02-04T13:22:50.042858Z","shell.execute_reply":"2025-02-04T13:22:50.049165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submit = pd.DataFrame({\n    'image_name' : data_test['image_name'],\n    'target' : prediction_\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T13:22:50.051413Z","iopub.execute_input":"2025-02-04T13:22:50.051665Z","iopub.status.idle":"2025-02-04T13:22:50.079482Z","shell.execute_reply.started":"2025-02-04T13:22:50.051617Z","shell.execute_reply":"2025-02-04T13:22:50.078855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submit.to_csv('submission.csv',index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T13:22:50.080369Z","iopub.execute_input":"2025-02-04T13:22:50.080675Z","iopub.status.idle":"2025-02-04T13:22:50.110941Z","shell.execute_reply.started":"2025-02-04T13:22:50.080627Z","shell.execute_reply":"2025-02-04T13:22:50.110344Z"}},"outputs":[],"execution_count":null}]}