{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport random\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import jaccard_score, confusion_matrix, ConfusionMatrixDisplay, classification_report\nfrom sklearn.utils import class_weight\nfrom sklearn.preprocessing import MinMaxScaler, RobustScaler\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n    \nimport tensorflow as tf\n\nfrom tensorflow.keras import Sequential,Model\nfrom tensorflow.keras.layers import Dense, Flatten, Dropout, Input, Concatenate, BatchNormalization, GlobalAveragePooling2D\n\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\nfrom tensorflow.keras.optimizers.schedules import ExponentialDecay\nfrom tensorflow.keras.utils import plot_model","metadata":{"execution":{"iopub.status.busy":"2021-12-01T11:07:19.652724Z","iopub.execute_input":"2021-12-01T11:07:19.652951Z","iopub.status.idle":"2021-12-01T11:07:19.660948Z","shell.execute_reply.started":"2021-12-01T11:07:19.652926Z","shell.execute_reply":"2021-12-01T11:07:19.659875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Directory for dataset\ntrain_csv_dir = \"/kaggle/input/siim-isic-melanoma-classification/train.csv\"\ntrain_jpeg_dir = \"/kaggle/input/siim-isic-melanoma-classification/jpeg/train/\"","metadata":{"execution":{"iopub.status.busy":"2021-12-01T11:07:19.662724Z","iopub.execute_input":"2021-12-01T11:07:19.663028Z","iopub.status.idle":"2021-12-01T11:07:19.674084Z","shell.execute_reply.started":"2021-12-01T11:07:19.662992Z","shell.execute_reply":"2021-12-01T11:07:19.673055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read datasets\ntrain_features = pd.read_csv(train_csv_dir)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T11:07:19.675253Z","iopub.execute_input":"2021-12-01T11:07:19.675564Z","iopub.status.idle":"2021-12-01T11:07:19.75112Z","shell.execute_reply.started":"2021-12-01T11:07:19.67553Z","shell.execute_reply":"2021-12-01T11:07:19.750099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_features.columns","metadata":{"execution":{"iopub.status.busy":"2021-12-01T11:07:19.752597Z","iopub.execute_input":"2021-12-01T11:07:19.752843Z","iopub.status.idle":"2021-12-01T11:07:19.75975Z","shell.execute_reply.started":"2021-12-01T11:07:19.752816Z","shell.execute_reply":"2021-12-01T11:07:19.759201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_features.info()","metadata":{"execution":{"iopub.status.busy":"2021-12-01T11:07:19.760811Z","iopub.execute_input":"2021-12-01T11:07:19.761337Z","iopub.status.idle":"2021-12-01T11:07:19.801873Z","shell.execute_reply.started":"2021-12-01T11:07:19.761303Z","shell.execute_reply":"2021-12-01T11:07:19.800852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's see the age histogram of malignant and benign cases\n# We will also see that the dataset is dramatically imbalanced\n\nbins = np.linspace(0, 100, 100)\nplt.hist((train_features['age_approx'][train_features['target']==1]), bins, alpha=1, label='malignant')\nplt.hist((train_features['age_approx'][train_features['target']==0]), bins, alpha=0.3, label='benign')\nplt.legend(loc='upper right')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-01T11:07:19.803186Z","iopub.execute_input":"2021-12-01T11:07:19.803703Z","iopub.status.idle":"2021-12-01T11:07:20.374501Z","shell.execute_reply.started":"2021-12-01T11:07:19.803661Z","shell.execute_reply":"2021-12-01T11:07:20.373927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sex features\ntrain_features['sex'] = train_features['sex'].map({'male': 1, 'female': 0})\ntrain_features['sex'] = train_features['sex'].fillna(-1)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T11:07:20.375819Z","iopub.execute_input":"2021-12-01T11:07:20.376256Z","iopub.status.idle":"2021-12-01T11:07:20.391332Z","shell.execute_reply.started":"2021-12-01T11:07:20.376225Z","shell.execute_reply":"2021-12-01T11:07:20.390532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Converting 'image_name' column for taking images\ntrain_features['image_name'] = train_features['image_name'].apply(lambda x : train_jpeg_dir + x + \".jpg\")","metadata":{"execution":{"iopub.status.busy":"2021-12-01T11:07:20.392637Z","iopub.execute_input":"2021-12-01T11:07:20.393315Z","iopub.status.idle":"2021-12-01T11:07:20.420143Z","shell.execute_reply.started":"2021-12-01T11:07:20.393282Z","shell.execute_reply":"2021-12-01T11:07:20.418826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# location features\nlocation_train = pd.get_dummies(train_features.anatom_site_general_challenge, prefix='loc')\ntrain_features = pd.concat((train_features[['image_name', 'age_approx', 'sex']], location_train, train_features[['target']]), axis = 1)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T11:07:20.421396Z","iopub.execute_input":"2021-12-01T11:07:20.421647Z","iopub.status.idle":"2021-12-01T11:07:20.439818Z","shell.execute_reply.started":"2021-12-01T11:07:20.421618Z","shell.execute_reply":"2021-12-01T11:07:20.438779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_features.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-01T11:07:20.44159Z","iopub.execute_input":"2021-12-01T11:07:20.442446Z","iopub.status.idle":"2021-12-01T11:07:20.45948Z","shell.execute_reply.started":"2021-12-01T11:07:20.442377Z","shell.execute_reply":"2021-12-01T11:07:20.45855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_features_malignant = np.copy(train_features[train_features['target']==1])\ntrain_features_benign = np.copy(train_features[train_features['target']==0])\n\ntrain_id_malignant = np.copy(train_features[train_features['target']==1])\ntrain_id_benign = np.copy(train_features[train_features['target']==0])\n\ntrain_features_benign_rest, train_features_benign_sample = train_test_split(train_features_benign,  \n                                                                            test_size = 0.25, random_state = 7, shuffle = True)\n","metadata":{"execution":{"iopub.status.busy":"2021-12-01T11:07:20.460531Z","iopub.execute_input":"2021-12-01T11:07:20.460745Z","iopub.status.idle":"2021-12-01T11:07:20.515178Z","shell.execute_reply.started":"2021-12-01T11:07:20.46072Z","shell.execute_reply":"2021-12-01T11:07:20.514473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# before age normalization, we split the dataset to avoid data leakage\ntrain_features_benign_sample, val_features_benign_sample= train_test_split(\n    train_features_benign_sample, test_size = 0.15, random_state = 7, shuffle = True)\n    \nval_features_benign_sample, testval_features_benign_sample = train_test_split(\n    val_features_benign_sample, test_size = 0.50, random_state = 7, shuffle = True )\n\ntrain_features_malignant, val_features_malignant= train_test_split(\n    train_features_malignant, test_size = 0.15, random_state = 7, shuffle = True)\n    \nval_features_malignant, testval_features_malignant = train_test_split(\n    val_features_malignant, test_size = 0.50, random_state = 7, shuffle = True )","metadata":{"execution":{"iopub.status.busy":"2021-12-01T11:07:20.518178Z","iopub.execute_input":"2021-12-01T11:07:20.518449Z","iopub.status.idle":"2021-12-01T11:07:20.532171Z","shell.execute_reply.started":"2021-12-01T11:07:20.518398Z","shell.execute_reply":"2021-12-01T11:07:20.531505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_features_sample = np.concatenate((train_features_benign_sample, train_features_malignant), axis = 0)\nval_features_sample = np.concatenate((val_features_benign_sample, val_features_malignant), axis = 0)\ntestval_features_sample = np.concatenate((testval_features_benign_sample, testval_features_malignant), axis = 0)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T11:07:20.533146Z","iopub.execute_input":"2021-12-01T11:07:20.534236Z","iopub.status.idle":"2021-12-01T11:07:20.540693Z","shell.execute_reply.started":"2021-12-01T11:07:20.534201Z","shell.execute_reply":"2021-12-01T11:07:20.539755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Age features after split\ndef robust_scaler(X):\n    scaler = RobustScaler()\n    age_scaled = scaler.fit_transform(X[:,1].reshape(-1, 1))\n    X[:,1] = age_scaled.reshape(-1,)\n    return X\n\ntrain = robust_scaler(train_features_sample)\nval = robust_scaler(val_features_sample)\ntestval = robust_scaler(testval_features_sample)","metadata":{"execution":{"iopub.status.busy":"2021-11-16T22:02:23.058451Z","iopub.execute_input":"2021-11-16T22:02:23.058711Z","iopub.status.idle":"2021-11-16T22:02:23.070826Z","shell.execute_reply.started":"2021-11-16T22:02:23.058675Z","shell.execute_reply":"2021-11-16T22:02:23.070039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DataGenerator(tf.keras.utils.Sequence):\n    def __init__(self, files, features, labels, batch_size, shuffle, random_state):\n        'Initialization'\n        self.files = files\n        self.features = features\n        self.labels = labels\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        self.random_state = random_state\n        self.on_epoch_end()\n    \n    \n    def __len__(self):\n        l = int(np.floor(len(self.files) / self.batch_size))\n        if l*self.batch_size < len(self.files):\n            l += 1\n        return l\n \n\n    def __getitem__(self, index):\n        # Generate indexes of the batch\n\n        y = self.labels[index*self.batch_size:(index+1)*self.batch_size]\n        f = self.features[index*self.batch_size:(index+1)*self.batch_size]\n\n        # Generate data\n        x = self.__data_generation(self.files[index*self.batch_size:(index+1)*self.batch_size])\n\n        return [np.array(x), np.array(f).astype('float32')], np.array(y).astype('int')\n\n    def on_epoch_end(self):\n        'Updates indexes after each epoch'\n        self.indexes = np.arange(len(self.files))\n        if self.shuffle == True:\n            np.random.seed(self.random_state)\n            np.random.shuffle(self.indexes)\n\n    \n    def __data_generation(self, files):\n             \n        imgs = []\n        def image_read(img_file):\n            image = tf.io.read_file(img_file)\n            image = tf.image.decode_jpeg(image, channels=3)\n            image = tf.cast(image, tf.float32)\n            image = tf.image.resize(image, (256, 256))\n            image = tf.keras.applications.efficientnet.preprocess_input(image) \n            return image\n\n        for img_file in files:\n            img = image_read(img_file)\n            imgs.append(img) \n        \n        return imgs","metadata":{"execution":{"iopub.status.busy":"2021-11-16T22:09:11.714881Z","iopub.execute_input":"2021-11-16T22:09:11.715136Z","iopub.status.idle":"2021-11-16T22:09:11.727587Z","shell.execute_reply.started":"2021-11-16T22:09:11.715107Z","shell.execute_reply":"2021-11-16T22:09:11.726904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create base model\nnoisy_student = \"../input/efficientnet-noisy-student-keras-applications/efficientnetb5_notop.h5\"\nefnet = tf.keras.applications.EfficientNetB5(\n    input_shape=(256, 256, 3), include_top=False)\nefnet.load_weights(noisy_student, by_name=True)\n# Freeze base model\nefnet.trainable = False\n\n","metadata":{"execution":{"iopub.status.busy":"2021-11-16T22:03:15.481268Z","iopub.execute_input":"2021-11-16T22:03:15.481941Z","iopub.status.idle":"2021-11-16T22:03:25.705047Z","shell.execute_reply.started":"2021-11-16T22:03:15.481904Z","shell.execute_reply":"2021-11-16T22:03:25.704271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model():\n    \n    input_image = Input(shape=(256, 256, 3))\n    input_feature = Input(shape=(8,))\n    \n    img_augmentation = Sequential([   \n    tf.keras.layers.experimental.preprocessing.RandomRotation(factor=(0.1, 0.2), fill_mode = 'reflect'),\n    tf.keras.layers.experimental.preprocessing.RandomTranslation(height_factor=0.1, width_factor=0.1, fill_mode = 'reflect'),\n    tf.keras.layers.experimental.preprocessing.RandomFlip(),\n    tf.keras.layers.experimental.preprocessing.RandomContrast(factor=(0.05, 0.15))],\n    name=\"img_augmentation\")\n    \n    x1 = img_augmentation(input_image)\n\n    # We make sure that the base_model is running in inference mode here by passing `training=False`. \n    x1 = efnet(x1, training=False)\n    x1 = tf.keras.layers.GlobalAveragePooling2D()(x1)\n\n    # Train the feature map with dense layers\n    x2 = tf.keras.layers.Dense(units = 64, activation=\"relu\")(input_feature)\n\n    # concatenate outputs of two models\n    x = tf.keras.layers.concatenate([x1, x2])\n    \n    # Train merged data with a dense layer\n    x = tf.keras.layers.Dense(units = 64, activation=\"relu\")(x)\n    x = tf.keras.layers.Dropout(0.4)(x)\n    \n    # make a prediction with single sigmoid unit\n    outputs = tf.keras.layers.Dense(units = 1, activation=\"sigmoid\")(x)\n    model = tf.keras.Model(inputs=[input_image, input_feature], outputs=outputs)\n   \n    lr_schedule = ExponentialDecay(\n        initial_learning_rate=initial_learning_rate,\n        decay_steps=100, decay_rate=0.9,\n        staircase=True)\n\n    # Compiling and Fitting the model\n    model.compile(loss='binary_crossentropy', \n              optimizer=tf.keras.optimizers.Adam(learning_rate = lr_schedule), \n              metrics=[tf.keras.metrics.AUC()])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2021-11-16T22:03:42.765777Z","iopub.execute_input":"2021-11-16T22:03:42.766399Z","iopub.status.idle":"2021-11-16T22:03:42.775126Z","shell.execute_reply.started":"2021-11-16T22:03:42.76636Z","shell.execute_reply":"2021-11-16T22:03:42.774043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"initial_learning_rate = 0.001\n\nmodel = build_model()\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-11-16T22:04:01.034757Z","iopub.execute_input":"2021-11-16T22:04:01.035374Z","iopub.status.idle":"2021-11-16T22:04:02.817018Z","shell.execute_reply.started":"2021-11-16T22:04:01.035335Z","shell.execute_reply":"2021-11-16T22:04:02.816336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_model(model, to_file='model_plot.png', show_shapes=True, show_layer_names=True)","metadata":{"execution":{"iopub.status.busy":"2021-11-15T13:31:55.070576Z","iopub.execute_input":"2021-11-15T13:31:55.070834Z","iopub.status.idle":"2021-11-15T13:31:55.967848Z","shell.execute_reply.started":"2021-11-15T13:31:55.070799Z","shell.execute_reply":"2021-11-15T13:31:55.965082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Early stopping helps as it stops training if val_loss(validation score) does not decrease.\nearly_stopping = EarlyStopping(patience = 5, mode='max', \n                              monitor='val_auc', restore_best_weights=True)\n\n# Class weights since the dataset is imbalanced \nclass_weight = {0: 0.07,\n               1: 0.93}\n\n# Reduce learning rate when there is no improvement on val_auc\nreduce_lr = ReduceLROnPlateau(monitor='val_auc', factor=0.2,\n                              patience=5, min_lr=0.01)\n\n# save the best model               \nbestmodel_path = './bestmodel/'\ncp_callback = ModelCheckpoint(filepath = bestmodel_path, mode='max', \n                              monitor='val_auc', \n                              verbose=2, save_best_only=True)\n","metadata":{"execution":{"iopub.status.busy":"2021-11-16T23:11:17.630214Z","iopub.execute_input":"2021-11-16T23:11:17.630918Z","iopub.status.idle":"2021-11-16T23:11:17.637245Z","shell.execute_reply.started":"2021-11-16T23:11:17.630881Z","shell.execute_reply":"2021-11-16T23:11:17.636104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train the model\nbatch_size=64\npredictor = model.fit(DataGenerator(files = train[:,0], \n                                    features = train[:,1:-1], \n                                    labels = train[:,-1], \n                                    batch_size = batch_size, shuffle=True, random_state=7),\n                      \n                      validation_data = (DataGenerator(files = val[:,0], \n                                                       features = val[:,1:-1], \n                                                       labels = val[:,-1], \n                                                       batch_size = batch_size, shuffle=False, random_state=7)),\n                               \n                      epochs=5, verbose = 1 #, class_weight = class_weight, callbacks = [early_stopping, reduce_lr, cp_callback])","metadata":{"execution":{"iopub.status.busy":"2021-11-16T23:11:22.516518Z","iopub.execute_input":"2021-11-16T23:11:22.516798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot loss during training\nplt.title('AUC')\nplt.plot(predictor.history['auc'], label='train')\nplt.plot(predictor.history['val_auc'], label='validation')\nplt.legend()\nplt.show()\n\n# plot loss during training\nplt.title('LOSS')\nplt.plot(predictor.history['loss'], label='train')\nplt.plot(predictor.history['val_loss'], label='validation')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-16T23:02:07.698024Z","iopub.execute_input":"2021-11-16T23:02:07.698633Z","iopub.status.idle":"2021-11-16T23:02:08.131683Z","shell.execute_reply.started":"2021-11-16T23:02:07.698595Z","shell.execute_reply":"2021-11-16T23:02:08.130924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_generetor = DataGenerator(files = testval[:,0], \n                                              features = testval[:,1:-1],\n                                              labels = testval[:,-1], \n                                              batch_size = 1, shuffle=False, random_state=7)","metadata":{"execution":{"iopub.status.busy":"2021-11-16T23:04:37.37663Z","iopub.execute_input":"2021-11-16T23:04:37.376902Z","iopub.status.idle":"2021-11-16T23:04:37.382578Z","shell.execute_reply.started":"2021-11-16T23:04:37.376875Z","shell.execute_reply":"2021-11-16T23:04:37.381655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction = model.predict_generator(test_generetor, len(testval))\ny_pred = [1 if x >= 0.5 else 0 for x in prediction]\nprint(classification_report(testval[:,-1].astype('int'), np.array(y_pred).astype('int')))","metadata":{"execution":{"iopub.status.busy":"2021-11-16T23:06:32.462702Z","iopub.execute_input":"2021-11-16T23:06:32.463667Z","iopub.status.idle":"2021-11-16T23:07:38.847814Z","shell.execute_reply.started":"2021-11-16T23:06:32.463621Z","shell.execute_reply":"2021-11-16T23:07:38.847008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm = confusion_matrix(testval[:,-1].astype('int'), np.array(y_pred).astype('int'))\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels = ['benign', 'malignant'])\ndisp.plot()","metadata":{"execution":{"iopub.status.busy":"2021-11-16T23:07:51.270744Z","iopub.execute_input":"2021-11-16T23:07:51.271214Z","iopub.status.idle":"2021-11-16T23:07:51.505764Z","shell.execute_reply.started":"2021-11-16T23:07:51.271174Z","shell.execute_reply":"2021-11-16T23:07:51.505072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('./model/')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-10-19T19:11:47.04339Z","iopub.execute_input":"2021-10-19T19:11:47.043887Z","iopub.status.idle":"2021-10-19T19:12:24.008779Z","shell.execute_reply.started":"2021-10-19T19:11:47.043845Z","shell.execute_reply":"2021-10-19T19:12:24.008026Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}