{"cells":[{"metadata":{},"cell_type":"markdown","source":"## Importing Libraries","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport os\n\nimport cv2\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nfrom mpl_toolkits.axes_grid1 import ImageGrid\nimport numpy as np\nfrom keras.utils import np_utils\nfrom keras import applications\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras import optimizers\nfrom keras.models import Sequential, Model \nfrom keras.layers import Dropout, Flatten, Dense, GlobalAveragePooling2D\nfrom keras.callbacks import ModelCheckpoint, LearningRateScheduler, TensorBoard, EarlyStopping\nimport tensorflow as tf\nfrom keras.optimizers import Adam\nfrom tensorflow.python.keras import backend as K\nfrom sklearn.model_selection import train_test_split\n\nfrom PIL import Image\nfrom mpl_toolkits.axes_grid1 import ImageGrid","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# List files available\nprint(os.listdir(\"../input/siim-isic-melanoma-classification\"))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')\ntest = pd.read_csv('../input/siim-isic-melanoma-classification/test.csv')\nsubmission=pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test.columns","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## EDA","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"### Missing Count","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"missing_col = ['sex','age_approx','anatom_site_general_challenge']\n\nfig, axes = plt.subplots(ncols = 2, figsize = (20,4),dpi = 100)\nsns.barplot(x= train[missing_col].isnull().sum().index , y= train[missing_col].isnull().sum().values, ax=axes[0])\nsns.barplot(x= test[missing_col].isnull().sum().index , y= test[missing_col].isnull().sum().values, ax=axes[1])\n\naxes[0].set_ylabel('Missing Value Count', size = 15, labelpad =20)\n\naxes[0].tick_params(axis ='x', labelsize = 15)\naxes[0].tick_params(axis='y', labelsize = 15)\n\naxes[1].tick_params(axis ='x', labelsize = 15)\naxes[1].tick_params(axis='y', labelsize = 15)\naxes[0].set_title('Training set', fontsize = 12)\naxes[1].set_title('Test set', fontsize  =12)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['target'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Count of Target","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"fig,axes = plt.subplots(ncols =1, figsize = (6,3), dpi = 100)\n\nsns.countplot(x = 'target', hue = 'target' , data=train)\n\nplt.tick_params(axis='x', labelsize=10)\nplt.tick_params(axis='y', labelsize=10)\naxes.set_xticklabels(['Benign(32542)', 'Melignant (584)'])\n\nplt.title('Number of examples')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data = train.groupby(['target','sex'])['benign_malignant'].count().to_frame().reset_index()\nax = sns.catplot(x='target',y= 'benign_malignant', hue='sex',data=data ,kind='bar')\nplt.xlabel(\"0: Benign, 1: Melignant\")\nplt.ylabel(\"Count of cases\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data = train.groupby(['sex','anatom_site_general_challenge'])['target'].count().to_frame().reset_index()\nax = sns.catplot(x='anatom_site_general_challenge',y= 'target', hue='sex',data=data ,kind='bar')\nplt.gcf().set_size_inches(10,4)\nplt.xlabel(\"Location of Image\")\nplt.ylabel(\"Count of Cases\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Visualizing Images","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"CATEGORIES = ['benign','malignant']\nNUM_CATEGORIES = len(CATEGORIES)\nSEED = 1987\ndata_dir = '../input/siim-isic-melanoma-classification/jpeg/'\ntrain_dir = data_dir+ 'train/'\ntest_dir = data_dir +'test/'\nsample_submission = pd.read_csv(os.path.join('../input/siim-isic-melanoma-classification', 'sample_submission.csv'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(1, figsize=(15, 10))\ngrid = ImageGrid(fig, 111, nrows_ncols=(NUM_CATEGORIES, 5), axes_pad=0.05)\ni = 0\nfor category_id, category in enumerate(CATEGORIES):\n    for filepath in train[train['benign_malignant'] == category]['image_name'].values[:5]:\n        ax = grid[i]\n        img = Image.open(\"../input/siim-isic-melanoma-classification/jpeg/train/\"+filepath+\".jpg\")\n        img = img.resize((240,240))\n        ax.imshow(img)\n        ax.axis('off')\n        if i % 5 == 5 - 1:\n            ax.text(250, 112, category, verticalalignment='center')\n        i += 1\nplt.show();","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Model Creation","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"model = applications.VGG19(weights = \"imagenet\", include_top=False, input_shape = (300, 300, 3))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Freezing starting layers so that weight of those layers are not required to be trained.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"for layer in model.layers[:3]:\n    layer.trainable = False","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Adding few more layers and final prediction layer.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"x = model.output\nx = Flatten()(x)\nx = Dense(1024, activation=\"relu\")(x)\nx = Dropout(0.5)(x)\nx = Dense(1024, activation=\"relu\")(x)\npredictions = Dense(1, activation=\"sigmoid\")(x) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Thanks to https://www.kaggle.com/ibtesama/siim-baseline-keras-vgg16\ndef focal_loss(alpha=0.25,gamma=2.0):\n    def focal_crossentropy(y_true, y_pred):\n        y_true = tf.dtypes.cast(y_true, tf.float64)\n        y_pred = tf.dtypes.cast(y_pred, tf.float64)\n        bce = K.binary_crossentropy(y_true, y_pred)\n        \n        y_pred = K.clip(y_pred, K.epsilon(), 1.- K.epsilon())\n        p_t = (y_true*y_pred) + ((1-y_true)*(1-y_pred))\n        \n        alpha_factor = 1\n        modulating_factor = 1\n\n        alpha_factor = y_true*alpha + ((1-alpha)*(1-y_true))\n        modulating_factor = K.pow((1-p_t), gamma)\n\n        # compute the final loss and return\n        return K.mean(alpha_factor*modulating_factor*bce, axis=-1)\n    return focal_crossentropy","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"opt = Adam(lr=1e-4)\nmodel_final = Model(inputs = model.input, outputs = predictions)\nmodel_final.compile(loss=focal_loss(), metrics=[tf.keras.metrics.AUC()],optimizer=opt)\n#model_final.compile(loss = focal_loss(), optimizer = optimizers.SGD(lr=0.00001, momentum=0.9), metrics=[\"accuracy\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model_final.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model_final.load_weights('../input/melanoma-eda-vgg-keras-starter/vgg16_1.h5')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Data formating","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"labels=[]\ndata=[]\nfor i in range(train.shape[0]):\n    data.append(train_dir + train['image_name'].iloc[i]+'.jpg')\n    labels.append(train['target'].iloc[i])\ndf=pd.DataFrame(data)\ndf.columns=['images']\ndf['target']=labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_data=[]\nfor i in range(test.shape[0]):\n    test_data.append(test_dir + test['image_name'].iloc[i]+'.jpg')\ndf_test=pd.DataFrame(test_data)\ndf_test.columns=['images']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nX_train, X_val, y_train, y_val = train_test_split(df['images'],df['target'], test_size=0.2, random_state=1234)\ntrain=pd.DataFrame(X_train)\ntrain.columns=['images']\ntrain['target']=y_train\n\nvalidation=pd.DataFrame(X_val)\nvalidation.columns=['images']\nvalidation['target']=y_val","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Data Augmentation","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"datagen = ImageDataGenerator(\n            rescale=1./255,\n            rotation_range=360.,\n            width_shift_range=0.3,\n            height_shift_range=0.3,\n            zoom_range=0.3,\n            horizontal_flip=True,\n            vertical_flip=True)\nval_datagen=ImageDataGenerator(rescale=1./255)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_generator = datagen.flow_from_dataframe(\n    train,\n    x_col='images',\n    y_col='target',\n    target_size=(300, 300),\n    batch_size=64,\n    shuffle=True,\n    class_mode='raw')\n\nvalidation_generator = val_datagen.flow_from_dataframe(\n    validation,\n    x_col='images',\n    y_col='target',\n    target_size=(300, 300),\n    shuffle=False,\n    batch_size=64,\n    class_mode='raw')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"checkpoint = ModelCheckpoint(\"vgg16_1.h5\", monitor='loss', verbose=1, save_best_only=True, save_weights_only=False, mode='auto', period=1)\nearly = EarlyStopping(monitor='loss', min_delta=0, patience=10, verbose=1, mode='auto')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"nb_epochs = 2\nbatch_size=64\nnb_train_steps = train.shape[0]//batch_size\nnb_val_steps=validation.shape[0]//batch_size\nprint(\"Number of training and validation steps: {} and {}\".format(nb_train_steps,nb_val_steps))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Training the model ","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"model_final.fit_generator(\n    train_generator,\n    epochs=nb_epochs,\n    validation_data=validation_generator,\n    callbacks=[checkpoint, early])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Submission","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"target=[]\nfor path in df_test['images']:\n    img=cv2.imread(str(path))\n    img = cv2.resize(img, (300,300))\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = img.astype(np.float32)/255.\n    img=np.reshape(img,(1,300,300,3))\n    prediction=model_final.predict(img)\n    target.append(prediction[0][0])\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission['target']=target","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)\nsubmission.head()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}