{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install -U efficientnet","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport cv2\nimport tensorflow as tf\nfrom tensorflow.keras.applications import *\nfrom tensorflow.keras.models import *\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.utils import Sequence\nfrom tensorflow.keras.callbacks import *\nfrom tensorflow.keras.optimizers import *\nimport efficientnet.tfkeras as efn\nimport matplotlib.pyplot as plt\nfrom kaggle_datasets import KaggleDatasets\nfrom sklearn.metrics import roc_auc_score\nfrom tensorflow.keras.metrics import AUC\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split,StratifiedKFold","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Set Path and Read DataFrames**","execution_count":null},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train_images_path='/kaggle/input/siim-isic-melanoma-classification/jpeg/train/'\ntest_images_path='/kaggle/input/siim-isic-melanoma-classification/jpeg/test/'\ntrain_df=pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/train.csv')\nsample_sub=pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Train Data Shape: {}'.format(train_df.shape))\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Image Ids check**\n* Checking if we have any duplicate id's of images","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"#Duplicate entries\nprint('Number of Unique ids: {}'.format(train_df['image_name'].nunique()))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Class Distribution**\n* Data is highly imbalanced ","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['target'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#target plotting\nzero_targets=train_df['target'][train_df['target']==0].count()\nones_targets=train_df['target'][train_df['target']==1].count()\nlabels=['Class 0','Class 1']\nt_circle=plt.Circle((0,0),0.7,color='white')\nplt.pie([zero_targets,ones_targets], labels=labels, colors=['red','green'])\np=plt.gcf()\np.gca().add_artist(t_circle)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Sneak peek at age relation to disease and sex**\n* We can see that in both males and females, people above 50 years have disease","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.catplot(x='sex',y='age_approx',data=train_df,hue='target')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Let's Plot some images**\n> Main Points:\n* images have varying shapes\n* Some images are brighter than others\n* In some images,Melanoma is not clearly visible\n* Apart from all above points, one main problem in images is presence of Hairs","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def read_images(ix):\n    img=cv2.imread(os.path.join(train_images_path,ix+'.jpg'))\n    img=cv2.cvtColor(img,cv2.COLOR_BGR2RGB)\n    return img","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"_,axs=plt.subplots(4,4,figsize=(13,13))\naxs=axs.flatten()\nfor img_ix,lbl,ax in zip(train_df['image_name'],train_df['target'],axs):\n    img=read_images(img_ix)\n    ax.imshow(img)\n    ax.set_title('Target: '.format(lbl))\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Modelling**","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"**TPU configurations**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Tensorflow version \" + tf.__version__)\nAUTO = tf.data.experimental.AUTOTUNE","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\nprint('Running on TPU ', tpu.master())\ntf.config.experimental_connect_to_cluster(tpu)\ntf.tpu.experimental.initialize_tpu_system(tpu)\n\nstrategy = tf.distribute.experimental.TPUStrategy(tpu)\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)\n\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Transform input data to TF dataset**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"gcs_path = KaggleDatasets().get_gcs_path()\ndef format_train_path(st):\n    return gcs_path + '/jpeg/train/' + st + '.jpg'\n\ndef format_test_path(st):\n    return gcs_path + '/jpeg/test/' + st + '.jpg'\n\ntrain_data,val_data=train_test_split(train_df,test_size=0.2)\n\ntrain_paths = train_data.image_name.apply(format_train_path).values\nval_paths = val_data.image_name.apply(format_train_path).values\n\ntrain_labels = train_data['target'].values\nval_labels = val_data['target'].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"DIMS=(512,512,3)\nEPOCHS=7","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def decode_image(filename,label=None,image_size=(DIMS[0],DIMS[1])):\n    bits=tf.io.read_file(filename)\n    img=tf.image.decode_jpeg(bits,channels=3)\n    img=tf.cast(img,tf.float32)/255.0\n    img=tf.image.resize(img,image_size)\n    if label is None:\n        return img\n    else:\n        return img, label\n    \ndef data_augment(image, label=None):\n    image = tf.image.random_flip_left_right(image)\n    image = tf.image.random_flip_up_down(image)\n    image = tf.image.adjust_brightness(image,0.2)\n    image = tf.image.rot90(image)\n    \n\n    if label is None:\n        return image\n    else:\n        return image, label","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_dataset=(tf.data.Dataset.from_tensor_slices((train_paths,train_labels)).map(decode_image,num_parallel_calls=AUTO)\n               .map(data_augment,num_parallel_calls=AUTO).repeat()\n              .shuffle(13)\n              .batch(BATCH_SIZE).prefetch(AUTO))\n\nval_dataset=(tf.data.Dataset.from_tensor_slices((val_paths,val_labels))\n             .map(decode_image,num_parallel_calls=AUTO)\n             .shuffle(13)\n             .batch(BATCH_SIZE)\n             .cache()\n             .prefetch(AUTO))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def focal_loss(gamma=2., alpha=.25):\n    def focal_loss_fixed(y_true, y_pred):\n        pt_1 = tf.where(tf.equal(y_true, 1), y_pred, tf.ones_like(y_pred))\n        pt_0 = tf.where(tf.equal(y_true, 0), y_pred, tf.zeros_like(y_pred))\n        return -K.mean(alpha * K.pow(1. - pt_1, gamma) * K.log(pt_1)) - K.mean((1 - alpha) * K.pow(pt_0, gamma) * K.log(1. - pt_0))\n    return focal_loss_fixed","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"with strategy.scope():\n    inp=Input(DIMS)\n    base_feat_0=efn.EfficientNetB7(weights='imagenet',include_top=False,input_tensor=inp)\n    base_feat_1=DenseNet169(weights='imagenet',include_top=False,input_tensor=inp)    \n    \n    x_0=GlobalAveragePooling2D()(base_feat_0.output)\n    x_1=GlobalAveragePooling2D()(base_feat_1.output)\n    x_1=Dense(2048)(x_1)\n    x_1=LeakyReLU()(x_1)\n    x_1=Dense(1024)(x_1)\n    x_1=LeakyReLU()(x_1)\n    \n    x=Concatenate()([x_0,x_1])\n    x=Dense(1024)(x)\n    x=LeakyReLU()(x)\n    \n    x=Dense(512)(x)\n    x=LeakyReLU()(x)\n    \n    out=Dense(1,activation='sigmoid')(x)\n    model=Model(inp,out)\n        \n    model.compile(\n        optimizer=Adam(),\n        loss = 'binary_crossentropy',\n        metrics=[AUC()]\n    )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"STEPS_PER_EPOCH = train_labels.shape[0] // BATCH_SIZE\nmc=ModelCheckpoint('classifier.h5',monitor='val_loss',save_best_only=True,verbose=1,period=1)\nrop=ReduceLROnPlateau(monitor='val_loss',min_lr=0.0000001,patience=2,mode='min')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history=model.fit(train_dataset,epochs=EPOCHS,steps_per_epoch=STEPS_PER_EPOCH,\n                  validation_data=val_dataset,\n                 callbacks=[mc,rop])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_metrics(metrics,name=['loss','Acc']):\n    epochs = range(1, len(metrics[0]) + 1)\n    plt.plot(epochs, metrics[0], 'b',color='red', label='Training '+name[0])\n    plt.plot(epochs, metrics[1], 'b',color='blue', label='Validation '+name[0])\n    plt.title('Metric Plot')\n    plt.legend()\n    plt.figure()\n    plt.plot(epochs, metrics[2], 'b', color='red', label='Training '+name[1])\n    plt.plot(epochs, metrics[3], 'b',color='blue', label='Validation '+name[1])\n    plt.legend()\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_metrics([history.history['loss'],history.history['val_loss'],\n              history.history['auc'],history.history['val_auc']])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Test Data**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"test_paths = sample_sub.image_name.apply(format_test_path).values\ntest_dataset=(tf.data.Dataset.from_tensor_slices(test_paths)\n             .map(decode_image,num_parallel_calls=AUTO)\n             .batch(BATCH_SIZE))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model=load_model('classifier.h5')\npreds=model.predict(test_dataset,verbose=1)\nsample_sub['target'] = preds\nsample_sub.to_csv('submission.csv', index=False)\nsample_sub.head()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}