{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"markdown","source":"This is a simple baseline model using VGG16. I have used the entire dataset to train this model. It gave me a public score of 0.853.\n","execution_count":null},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","collapsed":true,"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":false},"cell_type":"markdown","source":"#  Importing Libraries","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport keras\nfrom keras.preprocessing import image\nfrom keras.utils import to_categorical\n\nfrom keras.applications.vgg16 import VGG16,preprocess_input\nfrom keras.applications.resnet50 import ResNet50\nfrom keras.models import Sequential,Model\nfrom keras.layers import Conv2D, MaxPooling2D, Dense, Dropout, Input, Flatten,BatchNormalization,Activation\n\nfrom keras.applications.resnet50 import preprocess_input\nfrom keras.preprocessing.image import ImageDataGenerator\n\nfrom keras.optimizers import Adam, SGD, RMSprop\nfrom tensorflow.python.keras import backend as K\n\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\nimport cv2","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Exploring the data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Directory Listings for train and test images\n\ntrain_dir='/kaggle/input/siim-isic-melanoma-classification/jpeg/train/'\ntest_dir='/kaggle/input/siim-isic-melanoma-classification/jpeg/test/'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Reading the csv files\n\ntrain = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')\ntest = pd.read_csv('../input/siim-isic-melanoma-classification/test.csv')\nsubmission=pd.read_csv('../input/siim-isic-melanoma-classification/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Comparing the number of records in both categories\ntrain['target'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_samples = train.copy()\ntrain_samples.info()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Preparing the train and test data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Training data\ntrain_labels = []\ntrain_images =[]\n\nfor i in range(train_samples.shape[0]):\n    train_images.append(train_dir+train_samples['image_name'].iloc[i]+'.jpg')\n    train_labels.append(train_samples['target'].iloc[i])\n\ndf_train = pd.DataFrame(train_images)\ndf_train.columns =['images']\ndf_train['target'] = train_labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Test data\ntest_images =[]\nfor i in range(test.shape[0]):\n    test_images.append(test_dir+test['image_name'].iloc[i]+'.jpg')\n\ndf_test = pd.DataFrame(test_images)\ndf_test.columns = ['images']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Splitting the train data further into train and validation sets\nX_train, X_val, y_train,y_val = train_test_split(df_train['images'],df_train['target'],test_size=0.2,random_state=0)\n\ntrain = pd.DataFrame(X_train)\ntrain.columns = ['images']\ntrain['target']=y_train\n\nvalidation = pd.DataFrame(X_val)\nvalidation.columns = ['images']\nvalidation['target']=y_val","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Helper Functions","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_predictions(model,sub_df):\n    target=[]\n    for path in df_test['images']:\n        img=cv2.imread(str(path))\n        img = cv2.resize(img, (224,224))\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        img = img.astype(np.float32)/255.\n        img=np.reshape(img,(1,224,224,3))\n        prediction=model.predict(img)\n        target.append(prediction[0][0])\n    \n    sub_df['target']=target\n    return sub_df","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Data Preprocessing","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_datagen = ImageDataGenerator(preprocess_input,rescale=1./255,rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,horizontal_flip=True)\n\nval_datagen = ImageDataGenerator(preprocess_input,rescale=1./255)\n\nimage_size = 224\n\ntrain_generator = train_datagen.flow_from_dataframe(\n                    train,\n                    x_col='images',\n                    y_col ='target',\n                    target_size=(image_size,image_size),\n                    batch_size=8,\n                    shuffle=True,\n                    class_mode='raw')\n\nvalidation_generator = val_datagen.flow_from_dataframe(\n                    validation,\n                    x_col='images',\n                    y_col ='target',\n                    target_size=(image_size,image_size),\n                    batch_size=8,\n                    shuffle=False,\n                    class_mode='raw')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Modeling","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def vgg16_model(num_classes=None):\n    model = VGG16(weights='imagenet',include_top=False,input_shape=(224,224,3))\n    x = Flatten()(model.output)\n    output = Dense(1,activation='sigmoid')(x)\n    model = Model(model.input,output)\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"vgg_conv = vgg16_model(1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def focal_loss(alpha=0.25,gamma=2.0):\n    def focal_crossentropy(y_true, y_pred):\n        bce = K.binary_crossentropy(y_true, y_pred)\n        \n        y_pred = K.clip(y_pred, K.epsilon(), 1.- K.epsilon())\n        p_t = (y_true*y_pred) + ((1-y_true)*(1-y_pred))\n        \n        alpha_factor = 1\n        modulating_factor = 1\n\n        alpha_factor = y_true*alpha + ((1-alpha)*(1-y_true))\n        modulating_factor = K.pow((1-p_t), gamma)\n\n        # compute the final loss and return\n        return K.mean(alpha_factor*modulating_factor*bce, axis=-1)\n    return focal_crossentropy","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Defining the optimizer and compiling the model\nopt = Adam(lr=1e-5)\nvgg_conv.compile(loss=focal_loss(),optimizer=opt,metrics=[keras.metrics.AUC()])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Denining the num of epochs, batch_size and steps for training and validation\nnb_epochs = 2\nbatch_size=8\nnb_train_steps = train.shape[0]//batch_size  # // rounds off the result of division\nnb_validation_steps = validation.shape[0]//batch_size\nprint(\"Number of training and validation steps are {} and {}\".format(nb_train_steps,nb_validation_steps))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Fitting the model\nvgg_conv.fit_generator(\n    train_generator,\n    steps_per_epoch=nb_train_steps,\n    epochs=nb_epochs,\n    validation_data=validation_generator,\n    validation_steps=nb_validation_steps)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Getting the predictions for test data\n\nsub_vgg16 = submission.copy()\nsub_vgg16 = get_predictions(vgg_conv,sub_vgg16)\nsub_vgg16.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_vgg16.to_csv('submission_vgg16_Complete.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}