{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \n#from keras.preprocessing.image import ImageDataGenerator, load_img, img_to_array\nimport os   \n\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport cv2\nimport numpy as np\nimport tensorflow as tf\nimport tensorflow_datasets as tfds\n\nfrom tensorflow.keras import layers\nimport tensorflow as tf\nfrom tensorflow.keras.applications.inception_v3 import InceptionV3\n\nfrom keras.preprocessing.image import ImageDataGenerator\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-09T19:11:45.600210Z","iopub.execute_input":"2023-07-09T19:11:45.600599Z","iopub.status.idle":"2023-07-09T19:11:45.607612Z","shell.execute_reply.started":"2023-07-09T19:11:45.600565Z","shell.execute_reply":"2023-07-09T19:11:45.606385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/siim-isic-melanoma-classification/train.csv\")\npath_train = '/kaggle/input/siim-isic-melanoma-classification/jpeg/train'\ndirectory = '../input/siim-isic-melanoma-classification'","metadata":{"execution":{"iopub.status.busy":"2023-07-09T19:19:44.209759Z","iopub.execute_input":"2023-07-09T19:19:44.210184Z","iopub.status.idle":"2023-07-09T19:19:44.269106Z","shell.execute_reply.started":"2023-07-09T19:19:44.210152Z","shell.execute_reply":"2023-07-09T19:19:44.268156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"8th the data so that it runs faster","metadata":{}},{"cell_type":"code","source":"grouped = train_df.groupby('target')\nprint(len(grouped))\nnegGroup = grouped.get_group(0)\nposGroup = grouped.get_group(1)\nnegGroup = negGroup.sample(int(len(train_df)/8))","metadata":{"execution":{"iopub.status.busy":"2023-07-09T19:19:45.753600Z","iopub.execute_input":"2023-07-09T19:19:45.753960Z","iopub.status.idle":"2023-07-09T19:19:45.773285Z","shell.execute_reply.started":"2023-07-09T19:19:45.753930Z","shell.execute_reply":"2023-07-09T19:19:45.772270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_df = negGroup\ntrain_df = pd.concat([posGroup, negGroup])\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2023-07-09T19:19:47.881542Z","iopub.execute_input":"2023-07-09T19:19:47.881904Z","iopub.status.idle":"2023-07-09T19:19:47.905724Z","shell.execute_reply.started":"2023-07-09T19:19:47.881871Z","shell.execute_reply":"2023-07-09T19:19:47.904035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-07-09T19:19:49.829379Z","iopub.execute_input":"2023-07-09T19:19:49.829752Z","iopub.status.idle":"2023-07-09T19:19:49.838324Z","shell.execute_reply.started":"2023-07-09T19:19:49.829718Z","shell.execute_reply":"2023-07-09T19:19:49.837195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Getting Image Paths","metadata":{}},{"cell_type":"code","source":"#get data - images path prefix onto name\npath_train = directory + '/train/' + train_df['image_name'] + '.dcm'\n#path_test = directory + '/test/' + test_df['image_name'] + '.dcm'\n\ntrain_df['path_dicom'] = path_train\n#test_df['path_dicom'] = path_test\n\npath_train = directory + '/jpeg/train/' + train_df['image_name'] + '.jpg'\n#path_test = directory + '/jpeg/test/' + test_df['image_name'] + '.jpg'\n\ntrain_df['path_jpeg'] = path_train\n#test_df['path_jpeg'] = path_test","metadata":{"execution":{"iopub.status.busy":"2023-07-09T19:20:00.978631Z","iopub.execute_input":"2023-07-09T19:20:00.979061Z","iopub.status.idle":"2023-07-09T19:20:00.989572Z","shell.execute_reply.started":"2023-07-09T19:20:00.979021Z","shell.execute_reply":"2023-07-09T19:20:00.988597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Oversampling","metadata":{}},{"cell_type":"code","source":"train_df['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-07-09T19:20:00.991913Z","iopub.execute_input":"2023-07-09T19:20:00.993389Z","iopub.status.idle":"2023-07-09T19:20:01.006282Z","shell.execute_reply.started":"2023-07-09T19:20:00.993320Z","shell.execute_reply":"2023-07-09T19:20:01.005359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Test/Train Split","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain_df, test_df = train_test_split(train_df, test_size=0.2)","metadata":{"execution":{"iopub.status.busy":"2023-07-09T19:20:01.009080Z","iopub.execute_input":"2023-07-09T19:20:01.009540Z","iopub.status.idle":"2023-07-09T19:20:01.019550Z","shell.execute_reply.started":"2023-07-09T19:20:01.009508Z","shell.execute_reply":"2023-07-09T19:20:01.017611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_size = train_df['target'].value_counts().max()\n\nlst = [train_df]\nfor class_index, group in train_df.groupby('target'):\n    lst.append(group.sample(max_size-len(group), replace=True))\ntrain_df = pd.concat(lst)","metadata":{"execution":{"iopub.status.busy":"2023-07-09T19:20:01.020930Z","iopub.execute_input":"2023-07-09T19:20:01.021264Z","iopub.status.idle":"2023-07-09T19:20:01.035318Z","shell.execute_reply.started":"2023-07-09T19:20:01.021240Z","shell.execute_reply":"2023-07-09T19:20:01.034187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-07-09T19:20:01.038030Z","iopub.execute_input":"2023-07-09T19:20:01.038589Z","iopub.status.idle":"2023-07-09T19:20:01.047843Z","shell.execute_reply.started":"2023-07-09T19:20:01.038555Z","shell.execute_reply":"2023-07-09T19:20:01.046655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Adding images to dataset","metadata":{}},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(\n    rescale = 1. / 255.,\n    rotation_range = 180,\n    width_shift_range = 0.15,\n    height_shift_range = 0.15,\n    zoom_range = 0.1,\n    horizontal_flip = True,\n    vertical_flip = True,\n    brightness_range = [0.3,1.3],\n    fill_mode = 'reflect', # nearest\n    preprocessing_function = None,\n    validation_split = 0.2\n)\n\n#val_datagen = ImageDataGenerator(rescale = 1./255)","metadata":{"execution":{"iopub.status.busy":"2023-07-09T19:50:36.068335Z","iopub.execute_input":"2023-07-09T19:50:36.069226Z","iopub.status.idle":"2023-07-09T19:50:36.075404Z","shell.execute_reply.started":"2023-07-09T19:50:36.069179Z","shell.execute_reply":"2023-07-09T19:50:36.074207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['target'] = train_df['target'].astype(str)\n#test_df['target'] = test_df['target'].astype(str)","metadata":{"execution":{"iopub.status.busy":"2023-07-09T19:50:36.078467Z","iopub.execute_input":"2023-07-09T19:50:36.078824Z","iopub.status.idle":"2023-07-09T19:50:36.099174Z","shell.execute_reply.started":"2023-07-09T19:50:36.078788Z","shell.execute_reply":"2023-07-09T19:50:36.097945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_SHAPE = (224, 224, 3)\nBATCH_SIZE = 32\n\ntrain_generator = train_datagen.flow_from_dataframe(\n    train_df,\n    x_col = 'path_jpeg',\n    y_col = 'target',\n    subset = 'training',\n    target_size = (IMAGE_SHAPE[0], IMAGE_SHAPE[1]),\n    batch_size = BATCH_SIZE,\n    shuffle = True,\n    class_mode = 'binary')\n\n\nvalidation_generator = train_datagen.flow_from_dataframe(\n    train_df,\n    x_col = 'path_jpeg',\n    y_col = 'target',\n    subset = 'validation',\n    target_size = (IMAGE_SHAPE[0], IMAGE_SHAPE[1]),\n    shuffle = True,\n    batch_size = BATCH_SIZE,\n    class_mode='binary')","metadata":{"execution":{"iopub.status.busy":"2023-07-09T19:50:36.101251Z","iopub.execute_input":"2023-07-09T19:50:36.101892Z","iopub.status.idle":"2023-07-09T19:50:40.289373Z","shell.execute_reply.started":"2023-07-09T19:50:36.101858Z","shell.execute_reply":"2023-07-09T19:50:40.288440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Training data","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.models import Sequential\nfrom keras.layers import Input, Flatten\nfrom keras.layers import Dense\nfrom keras.optimizers import Adam\nfrom keras.layers import Dense, Input, Activation, add, Add, Dropout, BatchNormalization, GlobalAveragePooling2D, GlobalMaxPooling2D\nfrom tensorflow.keras import models\n\npre_model = EfficientNetB0(weights=\"imagenet\", include_top=False, input_shape=IMAGE_SHAPE)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-09T19:50:40.290867Z","iopub.execute_input":"2023-07-09T19:50:40.291268Z","iopub.status.idle":"2023-07-09T19:50:42.624136Z","shell.execute_reply.started":"2023-07-09T19:50:40.291229Z","shell.execute_reply":"2023-07-09T19:50:42.623168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = models.Sequential()\nmodel.add(pre_model)\nmodel.add(GlobalAveragePooling2D())\nmodel.add(Dense(1, activation='sigmoid'))\n#model.add(Dropout(0.2, name=\"dropout_out\"))\n#model.add(Flatten())\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-07-09T19:50:42.626369Z","iopub.execute_input":"2023-07-09T19:50:42.626659Z","iopub.status.idle":"2023-07-09T19:50:43.463639Z","shell.execute_reply.started":"2023-07-09T19:50:42.626633Z","shell.execute_reply":"2023-07-09T19:50:43.462699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-07-09T19:50:43.464911Z","iopub.execute_input":"2023-07-09T19:50:43.465861Z","iopub.status.idle":"2023-07-09T19:50:43.483329Z","shell.execute_reply.started":"2023-07-09T19:50:43.465825Z","shell.execute_reply":"2023-07-09T19:50:43.482317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_generator,\n    epochs = 10,\n    #steps_per_epoch = 300,\n    validation_data = validation_generator\n    #validation_steps = 7\n)","metadata":{"execution":{"iopub.status.busy":"2023-07-09T19:50:43.484579Z","iopub.execute_input":"2023-07-09T19:50:43.485010Z","iopub.status.idle":"2023-07-09T21:08:59.000241Z","shell.execute_reply.started":"2023-07-09T19:50:43.484949Z","shell.execute_reply":"2023-07-09T21:08:58.999290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nepochs = range(1,len(acc) + 1)\n\nplt.plot(epochs,acc,'bo',label = 'Training Accuracy')\nplt.plot(epochs,val_acc,'b',label = 'Validation Accuracy')\nplt.title('Training and Validation Accuracy')\nplt.legend()\nplt.figure()\n\nplt.plot(epochs,loss,'bo',label = 'Training loss')\nplt.plot(epochs,val_loss,'b',label = 'Validation Loss')\nplt.title('Training and Validation Loss')\nplt.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-09T21:08:59.001927Z","iopub.execute_input":"2023-07-09T21:08:59.002403Z","iopub.status.idle":"2023-07-09T21:08:59.600950Z","shell.execute_reply.started":"2023-07-09T21:08:59.002366Z","shell.execute_reply":"2023-07-09T21:08:59.599966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['target'] = test_df['target'].astype(str)\n\ntest_datagen = ImageDataGenerator(\n    rescale = 1. / 255.)\n\ntest_generator = test_datagen.flow_from_dataframe(\n    test_df,\n    x_col = 'path_jpeg',\n    y_col = 'target',\n    target_size = (IMAGE_SHAPE[0], IMAGE_SHAPE[1]),\n    batch_size = BATCH_SIZE,\n    shuffle = False,\n    class_mode = 'binary')","metadata":{"execution":{"iopub.status.busy":"2023-07-09T21:08:59.602497Z","iopub.execute_input":"2023-07-09T21:08:59.602844Z","iopub.status.idle":"2023-07-09T21:09:01.600041Z","shell.execute_reply.started":"2023-07-09T21:08:59.602810Z","shell.execute_reply":"2023-07-09T21:09:01.598306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#preds = np.argmax(model.predict(\n    #test_generator,\n    #steps=len(test_generator.filenames)), axis=-1)\npreds = model.predict(\n    test_generator)\n    #steps=len(test_generator.filenames))","metadata":{"execution":{"iopub.status.busy":"2023-07-09T21:09:01.604851Z","iopub.execute_input":"2023-07-09T21:09:01.605192Z","iopub.status.idle":"2023-07-09T21:10:15.312391Z","shell.execute_reply.started":"2023-07-09T21:09:01.605151Z","shell.execute_reply":"2023-07-09T21:10:15.311308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = preds.flatten()","metadata":{"execution":{"iopub.status.busy":"2023-07-09T21:10:15.314224Z","iopub.execute_input":"2023-07-09T21:10:15.314566Z","iopub.status.idle":"2023-07-09T21:10:15.318886Z","shell.execute_reply.started":"2023-07-09T21:10:15.314538Z","shell.execute_reply":"2023-07-09T21:10:15.317919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = y_pred.astype(int)","metadata":{"execution":{"iopub.status.busy":"2023-07-09T21:10:15.320555Z","iopub.execute_input":"2023-07-09T21:10:15.321223Z","iopub.status.idle":"2023-07-09T21:10:15.339165Z","shell.execute_reply.started":"2023-07-09T21:10:15.321184Z","shell.execute_reply":"2023-07-09T21:10:15.338248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true = test_df['target'].astype(int)\ny_true","metadata":{"execution":{"iopub.status.busy":"2023-07-09T21:10:15.362751Z","iopub.execute_input":"2023-07-09T21:10:15.363139Z","iopub.status.idle":"2023-07-09T21:10:15.375931Z","shell.execute_reply.started":"2023-07-09T21:10:15.363105Z","shell.execute_reply":"2023-07-09T21:10:15.375022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\n\n#y_true = test_df['target'].astype(int)\n\nlabels = [\"Negative\", \"Positive\"]\ncm = confusion_matrix(y_true, y_pred)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=labels)\ndisp.plot();\n    \n#confusion_matrix(y_true, y_pred, labels=[\"positive\", \"negative\"])","metadata":{"execution":{"iopub.status.busy":"2023-07-09T21:10:15.377401Z","iopub.execute_input":"2023-07-09T21:10:15.377975Z","iopub.status.idle":"2023-07-09T21:10:15.723797Z","shell.execute_reply.started":"2023-07-09T21:10:15.377942Z","shell.execute_reply":"2023-07-09T21:10:15.722906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_curve\nfrom sklearn.metrics import roc_auc_score\nfrom matplotlib import pyplot\n\nns_probs = [0 for _ in range(len(preds))]\n\nns_auc = roc_auc_score(y_true, ns_probs)\nlr_auc = roc_auc_score(y_true, y_pred)\n# summarize scores\nprint('No Skill: ROC AUC=%.3f' % (ns_auc))\nprint('Logistic: ROC AUC=%.3f' % (lr_auc))\n# calculate roc curves\nns_fpr, ns_tpr, _ = roc_curve(y_true, ns_probs)\nlr_fpr, lr_tpr, _ = roc_curve(y_true, y_pred)\n# plot the roc curve for the model\npyplot.plot(ns_fpr, ns_tpr, linestyle='--', label='No Skill')\npyplot.plot(lr_fpr, lr_tpr, marker='.', label='Logistic')\n# axis labels\npyplot.xlabel('False Positive Rate')\npyplot.ylabel('True Positive Rate')\n# show the legend\npyplot.legend()\n# show the plot\npyplot.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-09T21:10:15.725516Z","iopub.execute_input":"2023-07-09T21:10:15.726360Z","iopub.status.idle":"2023-07-09T21:10:16.003131Z","shell.execute_reply.started":"2023-07-09T21:10:15.726322Z","shell.execute_reply":"2023-07-09T21:10:16.002229Z"},"trusted":true},"execution_count":null,"outputs":[]}]}