{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"},{"sourceId":1200702,"sourceType":"datasetVersion","datasetId":680899}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom keras.layers import Input, Lambda, Dense, Flatten\nfrom keras.models import Model\nfrom keras.applications.vgg16 import VGG16\nfrom keras.applications.vgg16 import preprocess_input\nfrom keras.preprocessing import image\nfrom keras.models import Sequential\nfrom glob import glob\nimport matplotlib.pyplot as plt\n\nfrom keras.optimizers import Adam, SGD, RMSprop\nimport tensorflow as tf\nimport cv2\nimport glob\nfrom keras.preprocessing.image import load_img, img_to_array, array_to_img\nfrom tensorflow.python.keras import backend as K\nimport plotly.graph_objects as go\nimport plotly.offline as py\nautosize =False\n\nfrom plotly.subplots import make_subplots\nimport plotly.graph_objects as go\n\nimport pandas as pd\n\n%matplotlib inline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:50:24.469780Z","iopub.execute_input":"2024-12-07T11:50:24.470149Z","iopub.status.idle":"2024-12-07T11:50:24.478237Z","shell.execute_reply.started":"2024-12-07T11:50:24.470119Z","shell.execute_reply":"2024-12-07T11:50:24.477219Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dir='/kaggle/input/siim-isic-melanoma-classification/jpeg/train/'\ntest_dir='/kaggle/input/siim-isic-melanoma-classification/jpeg/test/'\ntrain=pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/train.csv')\ntest=pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:50:24.480037Z","iopub.execute_input":"2024-12-07T11:50:24.480881Z","iopub.status.idle":"2024-12-07T11:50:24.585374Z","shell.execute_reply.started":"2024-12-07T11:50:24.480841Z","shell.execute_reply":"2024-12-07T11:50:24.584617Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.target.value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:50:24.586366Z","iopub.execute_input":"2024-12-07T11:50:24.586615Z","iopub.status.idle":"2024-12-07T11:50:24.606269Z","shell.execute_reply.started":"2024-12-07T11:50:24.586592Z","shell.execute_reply":"2024-12-07T11:50:24.605376Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# function to draw bar plot\nimport matplotlib.pyplot as plt\ndef draw_bar_plot(category,length,xlabel,ylabel,title,sub):\n    plt.subplot(2,2,sub)\n    plt.bar(category, length)\n    plt.legend()\n    plt.xlabel(xlabel, fontsize=15)\n    plt.ylabel(ylabel, fontsize=15)\n    plt.title(title, fontsize=15)\n\nplt.figure(figsize = (8,6))\nplt.bar([\"Melanoma\",\"Normal\"],[len(train[train.target==1]), len(train[train.target==0])])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:50:24.608193Z","iopub.execute_input":"2024-12-07T11:50:24.608505Z","iopub.status.idle":"2024-12-07T11:50:24.844738Z","shell.execute_reply.started":"2024-12-07T11:50:24.608479Z","shell.execute_reply":"2024-12-07T11:50:24.843781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_benign=train[train['target']==0].sample(2000)\ndf_malignant=train[train['target']==1]\n\nprint('Benign Cases')\nbenign=[]\ndf_b=df_benign.head(30)\ndf_b=df_b.reset_index()\nfor i in range(30):\n    img=cv2.imread(str(train_dir + df_benign['image_name'].iloc[i]+'.jpg'))\n    img = cv2.resize(img, (224,224))\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = img.astype(np.float32)/255.\n    benign.append(img)\nf, ax = plt.subplots(5,6, figsize=(10,6))\nfor i, img in enumerate(benign):\n        ax[i//6, i%6].imshow(img)\n        ax[i//6, i%6].axis('off')\n        \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:50:24.845963Z","iopub.execute_input":"2024-12-07T11:50:24.846641Z","iopub.status.idle":"2024-12-07T11:50:29.067913Z","shell.execute_reply.started":"2024-12-07T11:50:24.846600Z","shell.execute_reply":"2024-12-07T11:50:29.066893Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Since this is a huge dataset, we would take a sample of it for training purpose\n\ndf_0=train[train['target']==0].sample(2000)\ndf_1=train[train['target']==1]\ntrain=pd.concat([df_0,df_1])\ntrain=train.reset_index()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:50:29.068957Z","iopub.execute_input":"2024-12-07T11:50:29.069252Z","iopub.status.idle":"2024-12-07T11:50:29.081765Z","shell.execute_reply.started":"2024-12-07T11:50:29.069225Z","shell.execute_reply":"2024-12-07T11:50:29.080961Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# update image names with the whole path\ndef append_ext(fn):\n    return train_dir+fn+\".jpg\"\ntrain[\"image_name\"]=train[\"image_name\"].apply(append_ext)\n\ndef append_ext(fn):\n    return test_dir+fn+\".jpg\"\ntest[\"image_name\"]=test[\"image_name\"].apply(append_ext)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:50:29.082920Z","iopub.execute_input":"2024-12-07T11:50:29.083187Z","iopub.status.idle":"2024-12-07T11:50:29.094416Z","shell.execute_reply.started":"2024-12-07T11:50:29.083163Z","shell.execute_reply":"2024-12-07T11:50:29.093377Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_val, y_train, y_val = train_test_split(train['image_name'],train['target'], test_size=0.2, random_state=1234)\n\ntrain=pd.DataFrame(X_train)\ntrain.columns=['image_name']\ntrain['target']=y_train\n\nvalidation=pd.DataFrame(X_val)\nvalidation.columns=['image_name']\nvalidation['target']=y_val","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:50:29.095668Z","iopub.execute_input":"2024-12-07T11:50:29.096058Z","iopub.status.idle":"2024-12-07T11:50:29.519897Z","shell.execute_reply.started":"2024-12-07T11:50:29.096000Z","shell.execute_reply":"2024-12-07T11:50:29.519219Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# resizing the images\nIMG_DIM = (224, 224)\n\n# load images\ntrain_imgs = [img_to_array(load_img(img, target_size=IMG_DIM)) for img in train.image_name]\ntrain_imgs = np.array(train_imgs)\n\nvalidation_imgs = [img_to_array(load_img(img, target_size=IMG_DIM)) for img in validation.image_name]\nvalidation_imgs = np.array(validation_imgs)\n\nprint('Train dataset shape:', train_imgs.shape, \n      '\\tValidation dataset shape:', validation_imgs.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:50:29.520822Z","iopub.execute_input":"2024-12-07T11:50:29.521334Z","iopub.status.idle":"2024-12-07T11:54:00.329450Z","shell.execute_reply.started":"2024-12-07T11:50:29.521306Z","shell.execute_reply":"2024-12-07T11:54:00.328507Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# define parameters for model training\nbatch_size = 128\nnum_classes = 2\nepochs = 30\ninput_shape = (224, 224, 3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:54:00.331963Z","iopub.execute_input":"2024-12-07T11:54:00.332290Z","iopub.status.idle":"2024-12-07T11:54:00.336253Z","shell.execute_reply.started":"2024-12-07T11:54:00.332262Z","shell.execute_reply":"2024-12-07T11:54:00.335405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def focal_loss(alpha=0.25,gamma=2.0):\n    def focal_crossentropy(y_true, y_pred):\n        bce = K.binary_crossentropy(y_true, y_pred)\n        \n        y_pred = K.clip(y_pred, K.epsilon(), 1.- K.epsilon())\n        p_t = (y_true*y_pred) + ((1-y_true)*(1-y_pred))\n        \n        alpha_factor = 1\n        modulating_factor = 1\n\n        alpha_factor = y_true*alpha + ((1-alpha)*(1-y_true))\n        modulating_factor = K.pow((1-p_t), gamma)\n\n        # compute the final loss and return\n        return K.mean(alpha_factor*modulating_factor*bce, axis=-1)\n    return focal_crossentropy","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:54:00.337222Z","iopub.execute_input":"2024-12-07T11:54:00.337461Z","iopub.status.idle":"2024-12-07T11:54:00.348500Z","shell.execute_reply.started":"2024-12-07T11:54:00.337438Z","shell.execute_reply":"2024-12-07T11:54:00.347795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch.optim as optim\n\n# we will use Adam optimizer\nopt = Adam(learning_rate=0.001)\n\n#total number of iterations is always equal to the total number of training samples divided by the batch_size.\nnb_train_steps = train.shape[0]//batch_size\nnb_val_steps=validation.shape[0]//batch_size\n\nprint(\"Number of training and validation steps: {} and {}\".format(nb_train_steps,nb_val_steps))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:54:00.349318Z","iopub.execute_input":"2024-12-07T11:54:00.349520Z","iopub.status.idle":"2024-12-07T11:54:03.999646Z","shell.execute_reply.started":"2024-12-07T11:54:00.349500Z","shell.execute_reply":"2024-12-07T11:54:03.998659Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n# Pixel Normalization and Image Augmentation\ntrain_datagen = ImageDataGenerator(rescale=1./255, zoom_range=0.3, rotation_range=50,\n                                   width_shift_range=0.2, height_shift_range=0.2, shear_range=0.2, \n                                   horizontal_flip=True, fill_mode='nearest')\n\n# no need to create augmentation images for validation data, only rescaling the pixels\nval_datagen = ImageDataGenerator(rescale=1./255)\n\ntrain_generator = train_datagen.flow(train_imgs, y_train, batch_size=batch_size)\nval_generator = val_datagen.flow(validation_imgs, y_val, batch_size=batch_size)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:54:04.000891Z","iopub.execute_input":"2024-12-07T11:54:04.001563Z","iopub.status.idle":"2024-12-07T11:54:04.062319Z","shell.execute_reply.started":"2024-12-07T11:54:04.001522Z","shell.execute_reply":"2024-12-07T11:54:04.061573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img_id = 55\ngenerator_100 = train_datagen.flow(train_imgs[img_id:img_id+1], train.target[img_id:img_id+1],\n                                   batch_size=1)\naug_img = [next(generator_100) for i in range(0,5)]\nfig, ax = plt.subplots(1,5, figsize=(16, 6))\nprint('Labels:', [item[1][0] for item in aug_img])\nl = [ax[i].imshow(aug_img[i][0][0]) for i in range(0,5)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:54:04.063414Z","iopub.execute_input":"2024-12-07T11:54:04.063772Z","iopub.status.idle":"2024-12-07T11:54:04.994644Z","shell.execute_reply.started":"2024-12-07T11:54:04.063736Z","shell.execute_reply":"2024-12-07T11:54:04.993670Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\ndel train\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:54:04.995725Z","iopub.execute_input":"2024-12-07T11:54:04.996093Z","iopub.status.idle":"2024-12-07T11:54:05.266700Z","shell.execute_reply.started":"2024-12-07T11:54:04.996065Z","shell.execute_reply":"2024-12-07T11:54:05.265783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from keras.applications import vgg16\nfrom keras.models import Model\nimport keras\n\nvgg = vgg16.VGG16(include_top=False, weights='imagenet', \n                                     input_shape=input_shape)\n\noutput = vgg.layers[-1].output\noutput = keras.layers.Flatten()(output)\nvgg_model = Model(vgg.input, output)\n\nvgg_model.trainable = False\nfor layer in vgg_model.layers:\n    layer.trainable = False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:54:05.267760Z","iopub.execute_input":"2024-12-07T11:54:05.268108Z","iopub.status.idle":"2024-12-07T11:54:05.921650Z","shell.execute_reply.started":"2024-12-07T11:54:05.268074Z","shell.execute_reply":"2024-12-07T11:54:05.920692Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"vgg_model.trainable = True\n\nset_trainable = False\nfor layer in vgg_model.layers:\n    if layer.name in ['block5_conv1', 'block4_conv1']:\n        set_trainable = True\n    if set_trainable:\n        layer.trainable = True\n    else:\n        layer.trainable = False\n        \nlayers = [(layer, layer.name, layer.trainable) for layer in vgg_model.layers]\npd.DataFrame(layers, columns=['Layer Type', 'Layer Name', 'Layer Trainable'])    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:54:05.923126Z","iopub.execute_input":"2024-12-07T11:54:05.923888Z","iopub.status.idle":"2024-12-07T11:54:05.940783Z","shell.execute_reply.started":"2024-12-07T11:54:05.923847Z","shell.execute_reply":"2024-12-07T11:54:05.940072Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, InputLayer\nfrom keras.models import Sequential\nfrom keras import optimizers\n\nmodel = Sequential()\nmodel.add(vgg_model)\nmodel.add(Dense(512, activation='relu', input_dim=input_shape))\nmodel.add(Dropout(0.3))\nmodel.add(Dense(512, activation='relu'))\nmodel.add(Dropout(0.3))\nmodel.add(Dense(1, activation='sigmoid'))\n\n\nmodel.compile(loss=focal_loss(), metrics=[tf.keras.metrics.AUC()],optimizer=opt)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:54:05.942117Z","iopub.execute_input":"2024-12-07T11:54:05.942384Z","iopub.status.idle":"2024-12-07T11:54:05.965622Z","shell.execute_reply.started":"2024-12-07T11:54:05.942360Z","shell.execute_reply":"2024-12-07T11:54:05.964766Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from keras.callbacks import EarlyStopping\nes = EarlyStopping(monitor='loss', patience=3, verbose=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:54:05.966833Z","iopub.execute_input":"2024-12-07T11:54:05.967220Z","iopub.status.idle":"2024-12-07T11:54:05.972222Z","shell.execute_reply.started":"2024-12-07T11:54:05.967182Z","shell.execute_reply":"2024-12-07T11:54:05.971360Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.fit(train_generator, steps_per_epoch=nb_train_steps, epochs=epochs,callbacks=[es],\n                              validation_data=val_generator, validation_steps=nb_val_steps, \n                              verbose=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:54:05.973159Z","iopub.execute_input":"2024-12-07T11:54:05.973401Z","iopub.status.idle":"2024-12-07T11:56:15.677409Z","shell.execute_reply.started":"2024-12-07T11:54:05.973379Z","shell.execute_reply":"2024-12-07T11:56:15.676536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_test = np.load('/kaggle/input/siimisic-melanoma-resized-images/x_test_224.npy')\nx_test = x_test.astype('float16')\ntest_imgs_scaled = x_test / 255\ndel x_test\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:56:15.678524Z","iopub.execute_input":"2024-12-07T11:56:15.678788Z","iopub.status.idle":"2024-12-07T11:56:48.600582Z","shell.execute_reply.started":"2024-12-07T11:56:15.678762Z","shell.execute_reply":"2024-12-07T11:56:48.599675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target=[]\ni = 0\nfor img in test_imgs_scaled:\n    img1=np.reshape(img,(1,224,224,3))\n    prediction=model.predict(img1)\n    i = i + 1\n    print(\"predicted image no.\",i)\n    target.append(prediction[0][0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:56:48.601843Z","iopub.execute_input":"2024-12-07T11:56:48.602577Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission file\nsub=pd.read_csv(\"../input/siim-isic-melanoma-classification/sample_submission.csv\")\nsub['target']=target\n#sub.to_csv('submission.csv', index=False)\nsub.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub.to_csv('submission.csv',index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}