{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Detecting Cancer Lesion | Efficientnet | Convolution Neural Network | Binary Classification 📷","metadata":{"_uuid":"a497a817-c08d-47f0-9360-783017b9395c","_cell_guid":"2d63c794-098a-49fc-af65-d9e4a8f6b56e","trusted":true}},{"cell_type":"markdown","source":"downloaded resized dataset [here](https://www.kaggle.com/cdeotte/jpeg-melanoma-256x256).","metadata":{}},{"cell_type":"markdown","source":"# Importing libraries","metadata":{"_uuid":"425b5915-9a9c-4317-a0e4-b8d36e064064","_cell_guid":"faf16ab1-c9c3-4d13-b8ca-8013ebfa5550","trusted":true}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import preprocessing,metrics\nfrom sklearn.utils import shuffle\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.layers import Conv2D, MaxPool2D, Dropout, Dense, Flatten, BatchNormalization\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator,load_img, img_to_array, array_to_img\nimport cv2\nfrom tqdm import tqdm\nimport os\nfrom PIL import Image\nimport gc\nfrom keras.callbacks import EarlyStopping\nfrom imblearn.over_sampling import RandomOverSampler\nfrom sklearn.metrics import classification_report\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\n\nimport tensorflow_addons as tfa\n\n\nprint(tf. __version__)\n%config Completer.use_jedi = False # makes auto completion work in notebook","metadata":{"_uuid":"735c2351-0b52-4a89-871c-a897ed865d7b","_cell_guid":"17b97765-a33b-49a4-a6e0-cf61281e0d39","collapsed":false,"scrolled":true,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-01T11:29:06.510885Z","iopub.execute_input":"2022-07-01T11:29:06.511349Z","iopub.status.idle":"2022-07-01T11:29:18.566915Z","shell.execute_reply.started":"2022-07-01T11:29:06.511261Z","shell.execute_reply":"2022-07-01T11:29:18.566030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## exploring the metadata csv file","metadata":{"_uuid":"203cab4c-baa2-4ec2-8dc1-2eb725f07586","_cell_guid":"27731f03-f240-450e-87a7-3f5ada8672c0","trusted":true}},{"cell_type":"code","source":"#update with the dataset in use\ndirectory = \"../input/jpeg-melanoma-256x256/\"\ntrain = pd.read_csv(directory + 'train.csv')\ntest = pd.read_csv(directory + 'test.csv')","metadata":{"_uuid":"4282dc04-e322-4920-863b-d7bc30cdfbe3","_cell_guid":"0d831f62-afb9-40c3-8a80-cc3868c3e52d","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-01T11:29:18.568402Z","iopub.execute_input":"2022-07-01T11:29:18.569752Z","iopub.status.idle":"2022-07-01T11:29:18.710192Z","shell.execute_reply.started":"2022-07-01T11:29:18.569706Z","shell.execute_reply":"2022-07-01T11:29:18.709174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head(5)","metadata":{"_uuid":"d6a46863-fe56-4635-9858-3e98b284d4b5","_cell_guid":"4123ad59-88cc-48e6-87f4-214b07e678d5","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-01T11:29:18.714021Z","iopub.execute_input":"2022-07-01T11:29:18.714482Z","iopub.status.idle":"2022-07-01T11:29:18.740443Z","shell.execute_reply.started":"2022-07-01T11:29:18.714445Z","shell.execute_reply":"2022-07-01T11:29:18.739249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Kaggle users reported some duplicate images in the dataset, which might impact the model, this code removes them\ndup = pd.read_csv(\"/kaggle/input/siim-list-of-duplicates/2020_Challenge_duplicates.csv\")\n\ndrop_idx_list = []\nfor dup_image in dup.ISIC_id_paired:\n    for idx,image in enumerate(train.image_name):\n        if image == dup_image:\n            drop_idx_list.append(idx)\n\nprint(\"no. of duplicates in training dataset:\",len(drop_idx_list))\n\ntrain.drop(drop_idx_list,inplace=True)\n\nprint(\"updated dimensions of the training dataset:\",train.shape)","metadata":{"_uuid":"bc11915e-e1ad-4c05-9e3c-c974a1523921","_cell_guid":"003c5aef-bb8c-41fa-894c-3dd4e9f015fd","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-01T11:29:18.743947Z","iopub.execute_input":"2022-07-01T11:29:18.744428Z","iopub.status.idle":"2022-07-01T11:29:23.831946Z","shell.execute_reply.started":"2022-07-01T11:29:18.744383Z","shell.execute_reply":"2022-07-01T11:29:23.830710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Explore the data","metadata":{"_uuid":"f63def60-571b-4553-be25-48d556d8b8f1","_cell_guid":"cd5a4508-678f-483c-b00b-7d1c6596b55e","execution":{"iopub.status.busy":"2022-03-03T12:57:57.603071Z","iopub.execute_input":"2022-03-03T12:57:57.603902Z","iopub.status.idle":"2022-03-03T12:57:57.62357Z","shell.execute_reply.started":"2022-03-03T12:57:57.603782Z","shell.execute_reply":"2022-03-03T12:57:57.622851Z"},"trusted":true}},{"cell_type":"code","source":"plt.rcParams['figure.figsize'] = (10,10)\ncompare = train[\"target\"].value_counts()\nprint(compare)\nlabels = ['benign','malignant']\nsizes = [compare[0],compare[1]]\nexplode = (0, 0.1)\nfig1, ax1 = plt.subplots()\nax1.pie(sizes, explode=explode, labels=labels, autopct='%0.1f%%',\n        shadow=False, startangle=-45)\nplt.show()","metadata":{"_uuid":"d3973eb6-04cb-47aa-b23f-626d691d9a32","_cell_guid":"e1de56c2-0109-4cab-bf22-0114a5aba9a6","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-01T11:29:23.833224Z","iopub.execute_input":"2022-07-01T11:29:23.833523Z","iopub.status.idle":"2022-07-01T11:29:24.026971Z","shell.execute_reply.started":"2022-07-01T11:29:23.833497Z","shell.execute_reply":"2022-07-01T11:29:24.025212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Heavily skewed data, $\\approx 2\\% $ of the data is malignant, We need to take this into consideration","metadata":{"_uuid":"ffac8f2e-6cad-4b41-862d-064ce4b29008","_cell_guid":"a0e29bbc-f0f2-4796-b345-49f89165b31b","trusted":true}},{"cell_type":"code","source":"df_benign=train[train['target']==0]\ndf_malignant=train[train['target']==1]","metadata":{"_uuid":"762eeb80-f052-42a8-b246-7dbd51a93d0c","_cell_guid":"8a4fa0a9-1510-4114-8a3f-522103484c10","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-01T11:29:24.033475Z","iopub.execute_input":"2022-07-01T11:29:24.034052Z","iopub.status.idle":"2022-07-01T11:29:24.050452Z","shell.execute_reply.started":"2022-07-01T11:29:24.034002Z","shell.execute_reply":"2022-07-01T11:29:24.049045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Benign Cases')\nbenign=[]\ndf_b=df_benign.head(30)\ndf_b=df_b.reset_index()\nfor i in range(30):\n    img = cv2.imread(directory + \"train/\" + df_benign['image_name'].iloc[i]+'.jpg')\n    img = cv2.resize(img, (224,224))\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = img.astype(np.float32)/255.\n    benign.append(img)\nf, ax = plt.subplots(5,6, figsize=(15,10))\nfor i, img in enumerate(benign):\n        ax[i//6, i%6].imshow(img)\n        ax[i//6, i%6].axis('off')\n        \nplt.show()","metadata":{"_uuid":"7bb13a66-ac10-43ef-a4b0-f69b84467b56","_cell_guid":"d9eef7d3-d64c-4d53-b781-afa567387939","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-01T11:29:24.052166Z","iopub.execute_input":"2022-07-01T11:29:24.053020Z","iopub.status.idle":"2022-07-01T11:29:26.087266Z","shell.execute_reply.started":"2022-07-01T11:29:24.052941Z","shell.execute_reply":"2022-07-01T11:29:26.086082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Malignant Cases')\nmalignant=[]\ndf_m=df_malignant.head(30)\ndf_m=df_m.reset_index()\nfor i in range(30):\n    img = cv2.imread(directory + \"train/\"+ df_m['image_name'].iloc[i]+'.jpg')\n    img = cv2.resize(img, (224,224))\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = img.astype(np.float32)/255.\n    malignant.append(img)\nf, ax = plt.subplots(5,6, figsize=(15,10))\nfor i, img in enumerate(malignant):\n        ax[i//6, i%6].imshow(img)\n        ax[i//6, i%6].axis('off')\n        \nplt.show()","metadata":{"_uuid":"85c01efc-259b-43dc-8f20-b0dd328efab4","_cell_guid":"65db2386-5e8e-424d-9d07-f7b687a6ecfe","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-01T11:29:26.088714Z","iopub.execute_input":"2022-07-01T11:29:26.089942Z","iopub.status.idle":"2022-07-01T11:29:27.826996Z","shell.execute_reply.started":"2022-07-01T11:29:26.089898Z","shell.execute_reply":"2022-07-01T11:29:27.825943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Setting the dataset","metadata":{"_uuid":"d87b2dd5-9360-46ea-b55b-2b5768ac80b6","_cell_guid":"da64e29f-3c82-4434-82d8-d4eec38ce49c","trusted":true}},{"cell_type":"code","source":"# define parameters for model training, edit before cell for image dimension \nIMG_DIM = 256 # for input reshape layer\nbatch_size = 512\nnum_classes = 2\nepochs = 30\nvalidation_split = 0.15","metadata":{"_uuid":"4c47e48b-7884-42d2-8ed4-c4dcfdaf7668","_cell_guid":"ad010f48-e8cc-4053-956a-9ba1e536a6b3","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-01T11:29:27.828468Z","iopub.execute_input":"2022-07-01T11:29:27.828953Z","iopub.status.idle":"2022-07-01T11:29:27.835750Z","shell.execute_reply.started":"2022-07-01T11:29:27.828914Z","shell.execute_reply":"2022-07-01T11:29:27.834598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Note: Accuracy is not a helpful metric for this task. You can have 99.8%+ accuracy on this task by predicting False all the time.\nNote: that the model is fit using a larger than default batch size, this is important to ensure that each batch has a decent chance of containing a few of the postivie (malignant) samples. If the batch size was too small, they would likely have no malignant case to learn from.\n\n[Tensorflow tutorial on imbalanced data](https://www.tensorflow.org/tutorials/structured_data/imbalanced_data#setup)","metadata":{}},{"cell_type":"code","source":"#loading images into tf.data.dataset\nfile_paths = train[\"image_name\"].values # need to add .jpg\nlabels = train[\"target\"].values\n#labels = to_categorical(labels,num_classes)\nx_train, x_val, y_train, y_val = train_test_split(file_paths,\n                                                  labels, \n                                                  test_size=validation_split,\n                                                  random_state=32,\n                                                  stratify = labels)\n\nds_train = tf.data.Dataset.from_tensor_slices(( 'train/'+ x_train + '.jpg', y_train))\nds_val = tf.data.Dataset.from_tensor_slices(( 'train/'+ x_val + '.jpg', y_val))\n\n#read image, reshape and normalize\ndef read_image(image_file, label):\n    image = tf.io.read_file(directory + image_file)\n    image = tf.image.decode_image(image, dtype=tf.float32, channels=3)\n    #image = tf.cast(image, tf.float32) / 255.0 # not needed for efficenetnet, read documentation\n    image = tf.reshape(image, [IMG_DIM, IMG_DIM, 3])\n\n    return image , label\n\n\n#data augmentation\ndef augment(image,label):\n    datagen = tf.keras.preprocessing.image.ImageDataGenerator(width_shift_range=0.01,\n                                                              height_shift_range=0.01,\n                                                              shear_range=0.01, \n                                                              rotation_range=15, \n                                                              zoom_range=0.01)\n    return image, label\n\nprint(\"number of training images = {}\".format(len(ds_train)), \"number of val images = {}\".format(len(ds_val)))\n\nAUTOTUNE = tf.data.experimental.AUTOTUNE\n\nds_train = ds_train.map(read_image, num_parallel_calls = AUTOTUNE).map(augment).batch(batch_size)\nds_train = ds_train.prefetch(AUTOTUNE)\n\nds_val = ds_val.map(read_image, num_parallel_calls = AUTOTUNE).batch(batch_size)\nds_val = ds_val.prefetch(AUTOTUNE)","metadata":{"_uuid":"0d4b4517-db61-488d-9a44-f2714bba4bc9","_cell_guid":"826ded24-8ea2-489b-a1df-b6038e1c1eb4","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-01T11:29:27.840315Z","iopub.execute_input":"2022-07-01T11:29:27.840684Z","iopub.status.idle":"2022-07-01T11:29:28.083849Z","shell.execute_reply.started":"2022-07-01T11:29:27.840651Z","shell.execute_reply":"2022-07-01T11:29:28.083092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model = tf.keras.applications.efficientnet.EfficientNetB3(\n            input_shape = (IMG_DIM, IMG_DIM, 3),\n            weights = 'imagenet',\n            include_top = False\n        )\nbase_model.trainable = False\n\ninputs = tf.keras.layers.Input(shape=(IMG_DIM, IMG_DIM, 3))\nx = base_model(inputs, training = False)\nx = tf.keras.layers.GlobalAveragePooling2D()(x)\nx = tf.keras.layers.Dense(512, activation= 'relu')(x)\nx = tf.keras.layers.Dropout(0.25)(x)\noutputs = tf.keras.layers.Dense(1, activation='sigmoid')(x)\n\nmodel = tf.keras.Model(inputs=inputs, outputs= outputs)\nmodel.summary()","metadata":{"_uuid":"458f918a-fa68-43fd-8b74-3758fe7c7060","_cell_guid":"66eb8b2e-881c-42a4-97e2-99616002ba6f","collapsed":false,"jupyter":{"outputs_hidden":false},"scrolled":true,"execution":{"iopub.status.busy":"2022-07-01T11:29:28.085199Z","iopub.execute_input":"2022-07-01T11:29:28.085519Z","iopub.status.idle":"2022-07-01T11:29:33.037987Z","shell.execute_reply.started":"2022-07-01T11:29:28.085490Z","shell.execute_reply":"2022-07-01T11:29:33.036700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#tf.keras.backend.clear_session()","metadata":{"execution":{"iopub.status.busy":"2022-07-01T11:29:33.039626Z","iopub.execute_input":"2022-07-01T11:29:33.039988Z","iopub.status.idle":"2022-07-01T11:29:33.045222Z","shell.execute_reply.started":"2022-07-01T11:29:33.039931Z","shell.execute_reply":"2022-07-01T11:29:33.044053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compile\noptimizer = tf.keras.optimizers.Adam(learning_rate=0.001)\n\n\nMETRICS = [keras.metrics.BinaryAccuracy(name='accuracy'), \n          keras.metrics.Precision(name='precision'),\n          keras.metrics.Recall(name='recall'),\n          keras.metrics.AUC(name='auc')]\n\nlossfunction = tfa.losses.SigmoidFocalCrossEntropy()\n#lossfunction = tf.keras.losses.BinaryCrossentropy() \n\n\nmodel.compile(\n    optimizer=optimizer, loss=lossfunction, metrics=METRICS)\n\nclass_weights = {0: 0.5,\n                 1: 20}\n\nes = EarlyStopping(monitor='val_auc', patience=2, verbose=1)\n\nmodel.summary()","metadata":{"_uuid":"6d5ca649-c02f-4736-8031-43a91ddc4439","_cell_guid":"a9a702af-06fc-4173-b362-43ac25b4c798","execution":{"iopub.status.busy":"2022-07-01T11:29:33.046557Z","iopub.execute_input":"2022-07-01T11:29:33.047012Z","iopub.status.idle":"2022-07-01T11:29:33.114024Z","shell.execute_reply.started":"2022-07-01T11:29:33.046946Z","shell.execute_reply":"2022-07-01T11:29:33.113153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hist = model.fit(ds_train,\n                 batch_size = batch_size,\n                 validation_data = ds_val,\n                 epochs=epochs,\n                 verbose=1,\n                 class_weight = class_weights, \n                 callbacks=[es])","metadata":{"execution":{"iopub.status.busy":"2022-07-01T11:29:33.115111Z","iopub.execute_input":"2022-07-01T11:29:33.115747Z","iopub.status.idle":"2022-07-01T11:30:29.286914Z","shell.execute_reply.started":"2022-07-01T11:29:33.115710Z","shell.execute_reply":"2022-07-01T11:30:29.285009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"___\n___","metadata":{"_uuid":"c8f9104c-705a-4f4e-9e2d-d43785cbd653","_cell_guid":"0d4cac2f-5d46-43da-99eb-261fd0f9e863","trusted":true}},{"cell_type":"code","source":"plt.plot(hist.history['accuracy'], color='r', label=\"train accuracy\")\nplt.plot(hist.history['val_accuracy'], color='b', label=\"validation accuracy\")\nplt.title(\"Test accuracy\")\nplt.xlabel(\"Number of Epochs\")\nplt.ylabel(\"Loss\")\nplt.legend()\nplt.show()","metadata":{"_uuid":"5155cfbf-3732-40fb-a6ba-39b0c6d6d931","_cell_guid":"be6081c3-e781-48df-8a82-e397a83c211f","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-01T11:30:29.288140Z","iopub.status.idle":"2022-07-01T11:30:29.288583Z","shell.execute_reply.started":"2022-07-01T11:30:29.288376Z","shell.execute_reply":"2022-07-01T11:30:29.288396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(hist.history['loss'], color='r', label=\"Train loss\")\nplt.plot(hist.history['val_loss'], color='b', label=\"validation loss\")\nplt.title(\"Test Loss\")\nplt.xlabel(\"Number of Epochs\")\nplt.ylabel(\"Loss\")\nplt.legend()\nplt.show()","metadata":{"_uuid":"037f0727-cccb-43e8-823c-bc76688c95c6","_cell_guid":"d662f340-b8b6-4586-9397-3707fb78244c","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-01T11:30:29.289920Z","iopub.status.idle":"2022-07-01T11:30:29.290339Z","shell.execute_reply.started":"2022-07-01T11:30:29.290151Z","shell.execute_reply":"2022-07-01T11:30:29.290170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(hist.history['auc'], color='r', label=\"Train auc\")\nplt.plot(hist.history['val_auc'], color='b', label=\"validation auc\")\nplt.title(\"auc\")\nplt.xlabel(\"Number of Epochs\")\nplt.ylabel(\"Loss\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-01T11:30:29.292667Z","iopub.status.idle":"2022-07-01T11:30:29.294028Z","shell.execute_reply.started":"2022-07-01T11:30:29.293780Z","shell.execute_reply":"2022-07-01T11:30:29.293804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_cm(labels, predictions, p=0.5):\n    cm = confusion_matrix(labels, predictions > p)\n    plt.figure(figsize=(10,10))\n    sns.heatmap(cm, annot=True, fmt=\"d\")\n    plt.title('Confusion matrix @{:.2f}'.format(p))\n    plt.ylabel('Actual label')\n    plt.xlabel('Predicted label')\n\n    print('(True postive): ', cm[0][0])\n    print(' (False Positives): ', cm[0][1])\n    print('(False Negatives): ', cm[1][0])\n    print(' (True Negative): ', cm[1][1])\n    print('total malignant: ', np.sum(cm[1]))\n\nval_prediction = model.predict(ds_val)\nplot_cm(y_val, val_prediction)","metadata":{"execution":{"iopub.status.busy":"2022-07-01T11:30:29.295183Z","iopub.status.idle":"2022-07-01T11:30:29.296137Z","shell.execute_reply.started":"2022-07-01T11:30:29.295892Z","shell.execute_reply":"2022-07-01T11:30:29.295914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_save_dir = ('./model_complete_effnetb3_50epochs') #.h5' rename saved model\n#model.save(model_save_dir + '.h5')","metadata":{"_uuid":"b671b1c1-fba8-42eb-8fb3-889eefef5fd8","_cell_guid":"01738298-93ca-4021-9f0b-4861be53dc60","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-01T11:30:29.297568Z","iopub.status.idle":"2022-07-01T11:30:29.297941Z","shell.execute_reply.started":"2022-07-01T11:30:29.297760Z","shell.execute_reply":"2022-07-01T11:30:29.297777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"_uuid":"fa674cde-b30c-44f3-a2c8-64438d526e02","_cell_guid":"eceb522d-29fa-4f3a-8d9f-7ac00afc8536","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-01T11:30:29.299025Z","iopub.status.idle":"2022-07-01T11:30:29.299395Z","shell.execute_reply.started":"2022-07-01T11:30:29.299215Z","shell.execute_reply":"2022-07-01T11:30:29.299232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"saved_model = tf.keras.models.load_model(model_save_dir + '.h5')","metadata":{"_uuid":"3033f45b-025e-436c-8c7c-6e64a018aa9e","_cell_guid":"a6048116-89e9-499d-9241-cf11de6c89ee","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-01T11:30:29.300638Z","iopub.status.idle":"2022-07-01T11:30:29.301073Z","shell.execute_reply.started":"2022-07-01T11:30:29.300845Z","shell.execute_reply":"2022-07-01T11:30:29.300863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfile_paths_test = test[\"image_name\"].values # need to add .jpg\n\nds_test = tf.data.Dataset.from_tensor_slices(('test/'+ file_paths_test + '.jpg'))\n\ndef read_image_test(image_file):\n    image = tf.io.read_file(directory + image_file)\n    image = tf.image.decode_image(image, dtype=tf.float32, channels=3)\n    #image = tf.cast(image, tf.float32) / 255.0\n    image = tf.reshape(image, [IMG_DIM, IMG_DIM, 3])\n\n    return image\n\nds_test = ds_test.map(read_image_test, num_parallel_calls = AUTOTUNE).batch(batch_size)\n\nIMG_DIM","metadata":{"_uuid":"14e2e362-0e3c-4721-885a-18cb6a9b5c9d","_cell_guid":"bc696b96-f882-49cb-a1f3-baf9290d8143","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-01T11:30:29.303452Z","iopub.status.idle":"2022-07-01T11:30:29.303879Z","shell.execute_reply.started":"2022-07-01T11:30:29.303678Z","shell.execute_reply":"2022-07-01T11:30:29.303697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(file_paths_test)","metadata":{"_uuid":"607e89ef-aecb-4d0a-8ce2-3cff32fdd098","_cell_guid":"c701f6b0-b085-4a86-9618-f775bb2cf732","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-01T11:30:29.305261Z","iopub.status.idle":"2022-07-01T11:30:29.305679Z","shell.execute_reply.started":"2022-07-01T11:30:29.305478Z","shell.execute_reply":"2022-07-01T11:30:29.305497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction=model.predict(ds_test)","metadata":{"_uuid":"30c59bd8-d449-4509-842a-1a63a98dc039","_cell_guid":"0a3a5008-7784-4909-9c05-92096bbf42fd","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-01T11:30:29.306917Z","iopub.status.idle":"2022-07-01T11:30:29.307365Z","shell.execute_reply.started":"2022-07-01T11:30:29.307161Z","shell.execute_reply":"2022-07-01T11:30:29.307193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction.shape","metadata":{"_uuid":"8cac1102-e732-465a-ab88-be7739c477e6","_cell_guid":"4687b2f3-9412-4c3d-9bf1-41fcacb7999d","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-01T11:30:29.309207Z","iopub.status.idle":"2022-07-01T11:30:29.309579Z","shell.execute_reply.started":"2022-07-01T11:30:29.309401Z","shell.execute_reply":"2022-07-01T11:30:29.309418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction = pd.DataFrame(prediction)\nprediction = prediction.idxmax(axis=1)\nprediction.shape","metadata":{"_uuid":"ca803444-8302-4a56-aa77-adb963d9819e","_cell_guid":"41d6cbb5-b1fa-4b2b-a55f-18b34bfa7952","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-01T11:30:29.311190Z","iopub.status.idle":"2022-07-01T11:30:29.311741Z","shell.execute_reply.started":"2022-07-01T11:30:29.311525Z","shell.execute_reply":"2022-07-01T11:30:29.311546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_results_pd = pd.read_csv(\"../input/jpeg-melanoma-256x256/sample_submission.csv\")\noutput_results_pd['target'] = prediction.ravel().tolist()","metadata":{"_uuid":"8baa8762-885c-48f0-a1e8-cd3977e64e3e","_cell_guid":"8528512d-8c0d-46f0-b875-9fba4ac87406","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-01T11:30:29.313026Z","iopub.status.idle":"2022-07-01T11:30:29.313405Z","shell.execute_reply.started":"2022-07-01T11:30:29.313223Z","shell.execute_reply":"2022-07-01T11:30:29.313240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_file = output_results_pd.to_csv(model_save_dir +\".csv\", index = False)","metadata":{"_uuid":"82a56583-68b2-4782-8fb2-4d930aae6aa0","_cell_guid":"4d42af1a-3765-49ef-b046-5e04ddfc2b8e","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-01T11:30:29.314667Z","iopub.status.idle":"2022-07-01T11:30:29.315068Z","shell.execute_reply.started":"2022-07-01T11:30:29.314851Z","shell.execute_reply":"2022-07-01T11:30:29.314869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Future Work**","metadata":{"_uuid":"7e5902da-bbf0-42bc-8631-e68448d1881d","_cell_guid":"b4881765-5496-4883-9c4b-564548ff656d","trusted":true}},{"cell_type":"markdown","source":"**conclusion:**\nthe model is suffering from extreamly under-represented class under fitting, which is appearnt as the accuracy of train and val is high however the precision and recall are quite low even when using focal loss based on the paper found [here](https://arxiv.org/abs/1708.02002) is implemented in the model \n\n**in progress:**\nimplemnt of class_weight for weighting the loss function (during training only). This can be useful to tell the model to \"pay more attention\" to samples from an under-represented class.","metadata":{}}]}