{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":20270,"databundleVersionId":1222630},{"sourceType":"datasetVersion","sourceId":1362834,"datasetId":791645,"databundleVersionId":1395299}],"dockerImageVersionId":30919,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport os\nimport cv2\nimport random\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report,roc_curve\n\nfrom tensorflow.keras.utils import Sequence\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import VGG16,ResNet50\nfrom tensorflow.keras import layers\nfrom tensorflow.keras import Input,Model\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import backend as K","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-08T17:12:12.449215Z","iopub.execute_input":"2025-03-08T17:12:12.449593Z","iopub.status.idle":"2025-03-08T17:12:12.454437Z","shell.execute_reply.started":"2025-03-08T17:12:12.449535Z","shell.execute_reply":"2025-03-08T17:12:12.453666Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DEVICE = \"TPU\"\nprint(\"connecting to TPU...\")\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    print(\"Could not connect to TPU\")\n    tpu = None\nif tpu:\n    try:\n        print(\"initializing  TPU ...\")\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        print(\"TPU initialized\")\n    except _:\n        print(\"failed to initialize TPU\")\nelse:\n    DEVICE = \"GPU\"\n\nif DEVICE != \"TPU\":\n    print(\"Using default strategy for CPU and single GPU\")\n    strategy = tf.distribute.get_strategy()\n\nif DEVICE == \"GPU\":\n    print(\"Num GPUs Available: \", len(tf.config.experimental.list_physical_devices('GPU')))\n    \nAUTO              = tf.data.experimental.AUTOTUNE\nREPLICAS          = strategy.num_replicas_in_sync\nprint(\"REPLICAS: %d\" % REPLICAS)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T18:21:15.800923Z","iopub.execute_input":"2025-03-08T18:21:15.801216Z","iopub.status.idle":"2025-03-08T18:21:15.809028Z","shell.execute_reply.started":"2025-03-08T18:21:15.801194Z","shell.execute_reply":"2025-03-08T18:21:15.808279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T17:12:12.501186Z","iopub.execute_input":"2025-03-08T17:12:12.501400Z","iopub.status.idle":"2025-03-08T17:12:12.570310Z","shell.execute_reply.started":"2025-03-08T17:12:12.501382Z","shell.execute_reply":"2025-03-08T17:12:12.569642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_path = []\n\npattern = '../input/siim-isic-melanoma-classification/jpeg/train'\nfor i in train['image_name'].values:\n    path = os.path.join(pattern,i)\n    path += '.jpg'\n    image_path.append(path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T17:12:12.571400Z","iopub.execute_input":"2025-03-08T17:12:12.571746Z","iopub.status.idle":"2025-03-08T17:12:12.617315Z","shell.execute_reply.started":"2025-03-08T17:12:12.571715Z","shell.execute_reply":"2025-03-08T17:12:12.616475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['image_path'] = image_path\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T17:12:12.618813Z","iopub.execute_input":"2025-03-08T17:12:12.619053Z","iopub.status.idle":"2025-03-08T17:12:12.631203Z","shell.execute_reply.started":"2025-03-08T17:12:12.619034Z","shell.execute_reply":"2025-03-08T17:12:12.630493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.countplot(train,x='target')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T17:12:12.632397Z","iopub.execute_input":"2025-03-08T17:12:12.632681Z","iopub.status.idle":"2025-03-08T17:12:12.972914Z","shell.execute_reply.started":"2025-03-08T17:12:12.632661Z","shell.execute_reply":"2025-03-08T17:12:12.972088Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['target'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T17:12:12.973717Z","iopub.execute_input":"2025-03-08T17:12:12.974023Z","iopub.status.idle":"2025-03-08T17:12:12.980108Z","shell.execute_reply.started":"2025-03-08T17:12:12.974001Z","shell.execute_reply":"2025-03-08T17:12:12.979414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_class_0 = train[train.target==0].sample(3000,random_state=42)\ndata_class_1 = train[train.target==1]\nnew_data = pd.concat([data_class_0,data_class_1])\nsns.countplot(new_data,x='target')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T17:12:12.980931Z","iopub.execute_input":"2025-03-08T17:12:12.981204Z","iopub.status.idle":"2025-03-08T17:12:13.123593Z","shell.execute_reply.started":"2025-03-08T17:12:12.981174Z","shell.execute_reply":"2025-03-08T17:12:13.122919Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['target'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T17:12:13.124405Z","iopub.execute_input":"2025-03-08T17:12:13.124732Z","iopub.status.idle":"2025-03-08T17:12:13.130714Z","shell.execute_reply.started":"2025-03-08T17:12:13.124702Z","shell.execute_reply":"2025-03-08T17:12:13.130056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Data(Sequence):\n    def __init__(self, image_path, target, batch_size, target_size=(224, 224), aug=None, shuffle=True, seed=42,**kwargs):\n        super().__init__()\n        self.image_path = np.array(image_path)\n        self.target = np.array(target)\n        self.batch_size = batch_size\n        self.target_size = target_size\n        self.aug = aug\n        self.shuffle = shuffle\n        self.seed = seed\n        np.random.seed(self.seed) \n        self.on_epoch_end()\n    \n    def __len__(self):\n        return int(np.ceil(len(self.image_path) / self.batch_size))  \n\n    def __getitem__(self, item):\n        batch_indices = self.indices[item * self.batch_size: (item + 1) * self.batch_size]\n        image_path_batch = [self.image_path[i] for i in batch_indices]\n        label_batch = [self.target[i] for i in batch_indices]\n        images = [self.load_data(i) for i in image_path_batch]\n\n        images = np.array(images)\n        label_batch = np.array(label_batch)\n        \n        if self.aug:\n            auged = self.aug.flow(images, label_batch, shuffle=False)  \n            images, label_batch = next(auged)\n\n        return images, label_batch\n\n    def on_epoch_end(self):\n        self.indices = np.arange(len(self.image_path))\n        if self.shuffle:\n            np.random.shuffle(self.indices)  \n\n    def load_data(self, image_path):\n        img = image.load_img(image_path,target_size=self.target_size) \n        img = image.img_to_array(img)\n        img /= 255.0\n        return img","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T17:12:13.132863Z","iopub.execute_input":"2025-03-08T17:12:13.133071Z","iopub.status.idle":"2025-03-08T17:12:13.146791Z","shell.execute_reply.started":"2025-03-08T17:12:13.133053Z","shell.execute_reply":"2025-03-08T17:12:13.145989Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_and_preprocess_train(image_path, label, target_size=(224, 224)):\n    img = tf.io.read_file(image_path)\n    img = tf.image.decode_jpeg(img, channels=3)\n    img = tf.image.resize(img, target_size)\n    img = tf.image.random_flip_left_right(img) \n    img = tf.image.random_flip_up_down(img)    \n    img = tf.image.random_crop(img, size=[target_size[0], target_size[1], 3])\n    img = img - [123.68, 116.78, 103.94] \n    img = img / [58.40, 57.12, 57.37] \n    \n    return img, label\n\ndef load_and_preprocess_valid(image_path, label, target_size=(224, 224)):\n    img = tf.io.read_file(image_path)\n    img = tf.image.decode_jpeg(img, channels=3)\n    img = tf.image.resize(img, target_size)\n    img = img - [123.68, 116.78, 103.94]  \n    img = img / [58.40, 57.12, 57.37] \n    \n    return img, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T17:12:13.147795Z","iopub.execute_input":"2025-03-08T17:12:13.148097Z","iopub.status.idle":"2025-03-08T17:12:13.163529Z","shell.execute_reply.started":"2025-03-08T17:12:13.148067Z","shell.execute_reply":"2025-03-08T17:12:13.162726Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(new_data.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T17:12:13.164331Z","iopub.execute_input":"2025-03-08T17:12:13.164556Z","iopub.status.idle":"2025-03-08T17:12:13.183012Z","shell.execute_reply.started":"2025-03-08T17:12:13.164527Z","shell.execute_reply":"2025-03-08T17:12:13.182248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"datagen = ImageDataGenerator(\n    rotation_range=30,  \n    width_shift_range=0.2,  \n    height_shift_range=0.2, \n    shear_range=0.2,     \n    zoom_range=0.2,  \n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\ndataset = Data(new_data['image_path'].values,new_data['target'].values,batch_size=32,target_size=(224,224),aug=datagen)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T17:12:13.183713Z","iopub.execute_input":"2025-03-08T17:12:13.184004Z","iopub.status.idle":"2025-03-08T17:12:13.195132Z","shell.execute_reply.started":"2025-03-08T17:12:13.183986Z","shell.execute_reply":"2025-03-08T17:12:13.194434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for images,labels in dataset:\n    fig, ax = plt.subplots(4,8,figsize=(12,6))\n    ax = ax.flatten()\n\n    for value,ax_i in enumerate(ax):\n        ax_i.imshow(images[value])\n        ax_i.set_title(labels[value])\n        ax_i.axis('off')\n    break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T17:12:13.195935Z","iopub.execute_input":"2025-03-08T17:12:13.196149Z","iopub.status.idle":"2025-03-08T17:12:18.617349Z","shell.execute_reply.started":"2025-03-08T17:12:13.196120Z","shell.execute_reply":"2025-03-08T17:12:18.616392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_model(lr=0.0001):\n    base_model = VGG16(input_shape=(224,224,3),include_top = False)\n\n    for layer in base_model.layers:\n        layer.trainable = False\n    \n    input = base_model.layers[-1].output\n    x = layers.GlobalAveragePooling2D()(input)\n    x = layers.Dense(512, activation = 'relu')(x)\n    output = layers.Dense(1, activation = 'sigmoid')(x)\n\n    model = Model(base_model.input,output)\n    print(model.summary())\n    model.compile(\n        #loss = 'binary_crossentropy',\n        loss = tf.keras.losses.BinaryFocalCrossentropy(alpha=0.25,gamma=2.0),\n        #binary_focal_loss(alpha=0.2,gamma=2),\n        #metrics=[keras.metrics.Recall()],\n        metrics = ['acc'],\n        optimizer = keras.optimizers.Adam(learning_rate = lr),\n    )\n    return model\n\nmodel = create_model()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T17:13:49.797417Z","iopub.execute_input":"2025-03-08T17:13:49.797765Z","iopub.status.idle":"2025-03-08T17:13:51.401798Z","shell.execute_reply.started":"2025-03-08T17:13:49.797737Z","shell.execute_reply":"2025-03-08T17:13:51.401072Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_train,x_valid,y_train,y_valid = train_test_split(new_data['image_path'],new_data['target'].values,\n                                                  test_size=0.1,\n                                                  random_state=42,\n                                                  stratify=new_data['target'].values)\n\nprint(x_train.shape,y_train.shape)\nprint(x_valid.shape,y_valid.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T17:14:18.406082Z","iopub.execute_input":"2025-03-08T17:14:18.406368Z","iopub.status.idle":"2025-03-08T17:14:18.415877Z","shell.execute_reply.started":"2025-03-08T17:14:18.406347Z","shell.execute_reply":"2025-03-08T17:14:18.415136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"batch_size = 32\n\ntrain_dataset = tf.data.Dataset.from_tensor_slices((x_train, y_train))\ntrain_dataset = train_dataset.map(lambda x, y: load_and_preprocess_train(x, y, target_size=(224, 224)), num_parallel_calls=tf.data.AUTOTUNE)\ntrain_dataset = train_dataset.batch(batch_size).prefetch(tf.data.AUTOTUNE)\n\nvalid_dataset = tf.data.Dataset.from_tensor_slices((x_valid, y_valid))\nvalid_dataset = valid_dataset.map(lambda x, y: load_and_preprocess_valid(x, y, target_size=(224, 224)), num_parallel_calls=tf.data.AUTOTUNE)\nvalid_dataset = valid_dataset.batch(batch_size).prefetch(tf.data.AUTOTUNE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T17:15:21.460635Z","iopub.execute_input":"2025-03-08T17:15:21.460944Z","iopub.status.idle":"2025-03-08T17:15:21.536386Z","shell.execute_reply.started":"2025-03-08T17:15:21.460921Z","shell.execute_reply":"2025-03-08T17:15:21.535753Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class_0_weight = len(y_train) / (2 * np.bincount(y_train)[0])\nclass_1_weight = len(y_train) / (2 * np.bincount(y_train)[1])\nclass_weight = {0: class_0_weight,1:class_1_weight}\nprint(class_weight)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T17:15:31.795115Z","iopub.execute_input":"2025-03-08T17:15:31.795412Z","iopub.status.idle":"2025-03-08T17:15:31.800343Z","shell.execute_reply.started":"2025-03-08T17:15:31.795389Z","shell.execute_reply":"2025-03-08T17:15:31.799611Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"epochs = 100\nfactor = 0.2\n\nearly_stopping = keras.callbacks.EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True)\nreduce_lr = keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor= factor, patience=3, min_lr=1e-7)\n\nH = model.fit(\n    train_dataset,#class_weight = class_weight,\n    validation_data = valid_dataset,\n    epochs = epochs,\n    callbacks = [early_stopping , reduce_lr]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T17:15:53.296784Z","iopub.execute_input":"2025-03-08T17:15:53.297080Z","iopub.status.idle":"2025-03-08T17:55:00.357385Z","shell.execute_reply.started":"2025-03-08T17:15:53.297057Z","shell.execute_reply":"2025-03-08T17:55:00.356630Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.save(\"melanoma_model.h5\")  # Saves in HDF5 format\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T17:58:09.156420Z","iopub.execute_input":"2025-03-08T17:58:09.156805Z","iopub.status.idle":"2025-03-08T17:58:09.287388Z","shell.execute_reply.started":"2025-03-08T17:58:09.156774Z","shell.execute_reply":"2025-03-08T17:58:09.286733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig , ax = plt.subplots(1,2, figsize=(10,6))\n\ntrain_acc = H.history['acc']\nval_acc = H.history['val_acc']\ntrain_loss = H.history['loss']\nval_loss = H.history['val_loss']\n\nnum_epoch = len(train_acc)\nax[0].plot(range(1,num_epoch+1) , train_acc , label = 'Train')\nax[0].plot(range(1,num_epoch+1) , val_acc, label = 'Val')\nax[0].set_xlabel('Epochs')\nax[0].set_ylabel('Accuracy')\nax[0].legend()\n\nax[1].plot(range(1,num_epoch+1) , train_loss , label = 'Train')\nax[1].plot(range(1,num_epoch+1) , val_loss, label = 'Val')\nax[1].set_xlabel('Epochs')\nax[1].set_ylabel('Loss')\nax[1].legend()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T17:59:44.001799Z","iopub.execute_input":"2025-03-08T17:59:44.002133Z","iopub.status.idle":"2025-03-08T17:59:44.371202Z","shell.execute_reply.started":"2025-03-08T17:59:44.002108Z","shell.execute_reply":"2025-03-08T17:59:44.370234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = model.predict(valid_dataset)\ny_pred = np.where(y_pred>0.2,1,0)\n\nreport = classification_report(y_valid,y_pred)\nprint(report)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T18:00:31.748359Z","iopub.execute_input":"2025-03-08T18:00:31.748722Z","iopub.status.idle":"2025-03-08T18:00:41.552135Z","shell.execute_reply.started":"2025-03-08T18:00:31.748690Z","shell.execute_reply":"2025-03-08T18:00:41.551394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = pd.read_csv('../input/siim-isic-melanoma-classification/test.csv')\ntest.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T18:01:58.979870Z","iopub.execute_input":"2025-03-08T18:01:58.980177Z","iopub.status.idle":"2025-03-08T18:01:59.001327Z","shell.execute_reply.started":"2025-03-08T18:01:58.980152Z","shell.execute_reply":"2025-03-08T18:01:59.000630Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_path = []\n\npattern = '../input/siim-isic-melanoma-classification/jpeg/test'\nfor i in test['image_name'].values:\n    path = os.path.join(pattern,i)\n    path += '.jpg'\n    image_path.append(path)\n\ntest['image_path'] = image_path","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T18:02:33.247488Z","iopub.execute_input":"2025-03-08T18:02:33.247876Z","iopub.status.idle":"2025-03-08T18:02:33.267742Z","shell.execute_reply.started":"2025-03-08T18:02:33.247846Z","shell.execute_reply":"2025-03-08T18:02:33.266991Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"valid_dataset = tf.data.Dataset.from_tensor_slices((test['image_path'].values, None))\nvalid_dataset = valid_dataset.map(lambda x, _: load_and_preprocess_valid(x, _, target_size=(224, 224)), num_parallel_calls=tf.data.AUTOTUNE)\nvalid_dataset = valid_dataset.batch(batch_size).prefetch(tf.data.AUTOTUNE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T18:03:01.894377Z","iopub.execute_input":"2025-03-08T18:03:01.894686Z","iopub.status.idle":"2025-03-08T18:03:01.918680Z","shell.execute_reply.started":"2025-03-08T18:03:01.894662Z","shell.execute_reply":"2025-03-08T18:03:01.917985Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"prediction = model.predict(valid_dataset)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T18:03:08.955141Z","iopub.execute_input":"2025-03-08T18:03:08.955430Z","iopub.status.idle":"2025-03-08T18:07:46.804620Z","shell.execute_reply.started":"2025-03-08T18:03:08.955407Z","shell.execute_reply":"2025-03-08T18:07:46.803887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred = [i[0] for i in prediction]\nsubmit = pd.DataFrame({\n    'image_name' : test['image_name'],\n    'target' : pred\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T18:10:10.442245Z","iopub.execute_input":"2025-03-08T18:10:10.442542Z","iopub.status.idle":"2025-03-08T18:10:10.452316Z","shell.execute_reply.started":"2025-03-08T18:10:10.442520Z","shell.execute_reply":"2025-03-08T18:10:10.451358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submit.to_csv('submission.csv',index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T18:10:25.138113Z","iopub.execute_input":"2025-03-08T18:10:25.138400Z","iopub.status.idle":"2025-03-08T18:10:25.162297Z","shell.execute_reply.started":"2025-03-08T18:10:25.138378Z","shell.execute_reply":"2025-03-08T18:10:25.161660Z"}},"outputs":[],"execution_count":null}]}