{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install keras-tuner","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-21T08:00:20.252296Z","iopub.execute_input":"2023-06-21T08:00:20.252715Z","iopub.status.idle":"2023-06-21T08:00:25.874633Z","shell.execute_reply.started":"2023-06-21T08:00:20.252685Z","shell.execute_reply":"2023-06-21T08:00:25.873589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install keras-tuner\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Flatten, Conv2D, MaxPool2D, Dropout, BatchNormalization\nfrom tensorflow.keras.optimizers import RMSprop\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tqdm import tqdm\nfrom tensorflow.keras.preprocessing import image\nfrom kerastuner import RandomSearch","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:00:27.901431Z","iopub.execute_input":"2023-06-21T08:00:27.901800Z","iopub.status.idle":"2023-06-21T08:01:13.993514Z","shell.execute_reply.started":"2023-06-21T08:00:27.901767Z","shell.execute_reply":"2023-06-21T08:01:13.992485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Load the ground truth data\ntrain_data = pd.read_csv('/kaggle/input/isicgt/ISIC-2017_Training_Part3_GroundTruth.csv')\ntrain_data.head()\n\n# Load the training images\nstore_list = []\nimage_height = 350\nimage_width = 350\nfor i in tqdm(range(train_data.shape[0])):\n    path = '/kaggle/input/isic2019/ISIC-2017_Training_Data/' + train_data['image_id'][i] + '.jpg'\n    image_check = image.load_img(path, target_size=(image_height, image_width))\n    image_check = image.img_to_array(image_check)\n    # scaling the images\n    image_check = image_check/255\n    store_list.append(image_check)\nx_train = np.array(store_list)\ny_train = train_data.drop(columns=['image_id'])\ny_train = y_train.to_numpy()\n\n# Load the test data\ntest_data = pd.read_csv('/kaggle/input/test2017isic/ISIC-2017_Test_v2_Part3_GroundTruth.csv')\ntest_data.head()\n\n# Load the test images\nstore_list = []\nfor i in tqdm(range(test_data.shape[0])):\n    path = '/kaggle/input/test2017isic/ISIC-2017_Test_v2_Data/ISIC-2017_Test_v2_Data/' + test_data['image_id'][i] + '.jpg'\n    image_check = image.load_img(path, target_size=(image_height, image_width))\n    image_check = image.img_to_array(image_check)\n    # scaling the images\n    image_check = image_check/255\n    store_list.append(image_check)\nx_test = np.array(store_list)\ny_test = test_data.drop(columns=['image_id'])\ny_test = y_test.to_numpy()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:01:16.987314Z","iopub.execute_input":"2023-06-21T08:01:16.988383Z","iopub.status.idle":"2023-06-21T08:09:01.675200Z","shell.execute_reply.started":"2023-06-21T08:01:16.988347Z","shell.execute_reply":"2023-06-21T08:09:01.674035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Load the ground truth data for the training set\ntrain_data = pd.read_csv('/kaggle/input/isicgt/ISIC-2017_Training_Part3_GroundTruth.csv')\n\n# Count the number of images in each category for the training set\nnum_melanoma_train = (train_data['melanoma'] == 1).sum()\nnum_seborrheic_keratosis_train = (train_data['seborrheic_keratosis'] == 1).sum()\n\nprint(\"Number of melanoma images in the training set:\", num_melanoma_train)\nprint(\"Number of seborrheic keratosis images in the training set:\", num_seborrheic_keratosis_train)\n\n# Load the ground truth data for the test set\ntest_data = pd.read_csv('/kaggle/input/test2017isic/ISIC-2017_Test_v2_Part3_GroundTruth.csv')\n\n# Count the number of images in each category for the test set\nnum_melanoma_test = (test_data['melanoma'] == 1).sum()\nnum_seborrheic_keratosis_test = (test_data['seborrheic_keratosis'] == 1).sum()\n\nprint(\"Number of melanoma images in the test set:\", num_melanoma_test)\nprint(\"Number of seborrheic keratosis images in the test set:\", num_seborrheic_keratosis_test)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:10:32.590903Z","iopub.execute_input":"2023-06-21T08:10:32.591435Z","iopub.status.idle":"2023-06-21T08:10:32.610905Z","shell.execute_reply.started":"2023-06-21T08:10:32.591396Z","shell.execute_reply":"2023-06-21T08:10:32.609909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Load the ground truth data for the training set\ntrain_data = pd.read_csv('/kaggle/input/isicgt/ISIC-2017_Training_Part3_GroundTruth.csv')\n\n# Count the number of rows in the training data\nnum_train_images = train_data.shape[0]\n\nprint(\"Number of images in the training set:\", num_train_images)\n\n# Load the ground truth data for the test set\ntest_data = pd.read_csv('/kaggle/input/test2017isic/ISIC-2017_Test_v2_Part3_GroundTruth.csv')\n\n# Count the number of rows in the test data\nnum_test_images = test_data.shape[0]\n\nprint(\"Number of images in the test set:\", num_test_images)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:10:36.310386Z","iopub.execute_input":"2023-06-21T08:10:36.310884Z","iopub.status.idle":"2023-06-21T08:10:36.324371Z","shell.execute_reply.started":"2023-06-21T08:10:36.310848Z","shell.execute_reply":"2023-06-21T08:10:36.323518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping\n\ndatagen = ImageDataGenerator(\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest')\n\ndatagen.fit(x_train)\n\n# Define the EarlyStopping callback\nes_callback = EarlyStopping(monitor='val_loss', patience=3)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:10:41.173370Z","iopub.execute_input":"2023-06-21T08:10:41.173809Z","iopub.status.idle":"2023-06-21T08:10:42.232566Z","shell.execute_reply.started":"2023-06-21T08:10:41.173778Z","shell.execute_reply":"2023-06-21T08:10:42.231526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set the batch size and number of steps per epoch\nbatch_size = 32\nsteps_per_epoch_train = len(x_train) // batch_size\nsteps_per_epoch_test = len(x_test) // batch_size\n\n# Generate augmented images for the training set\ntrain_datagen = datagen.flow(x_train, y_train, batch_size=batch_size, shuffle=True)\nnum_augmented_images_train = steps_per_epoch_train * batch_size\n\n# Generate augmented images for the test set\ntest_datagen = datagen.flow(x_test, y_test, batch_size=batch_size, shuffle=True)\nnum_augmented_images_test = steps_per_epoch_test * batch_size\n\nprint(\"Number of augmented images in the training set:\", num_augmented_images_train)\nprint(\"Number of augmented images in the test set:\", num_augmented_images_test)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:10:46.668780Z","iopub.execute_input":"2023-06-21T08:10:46.669214Z","iopub.status.idle":"2023-06-21T08:10:46.677203Z","shell.execute_reply.started":"2023-06-21T08:10:46.669181Z","shell.execute_reply":"2023-06-21T08:10:46.676411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set the batch size\nbatch_size = 32\n\n# Generate augmented images\naugmented_images = datagen.flow(x_train, batch_size=batch_size, shuffle=True)\n\n# Determine the total number of images generated\nnum_augmented_images = batch_size * 10\n\nprint(\"Number of augmented images:\", num_augmented_images)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:10:50.557860Z","iopub.execute_input":"2023-06-21T08:10:50.558889Z","iopub.status.idle":"2023-06-21T08:10:50.564691Z","shell.execute_reply.started":"2023-06-21T08:10:50.558847Z","shell.execute_reply":"2023-06-21T08:10:50.563758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import keras\nbest_model = keras.models.load_model('/kaggle/input/best-model/isic2017.h5')","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:10:54.757765Z","iopub.execute_input":"2023-06-21T08:10:54.758211Z","iopub.status.idle":"2023-06-21T08:11:04.697362Z","shell.execute_reply.started":"2023-06-21T08:10:54.758176Z","shell.execute_reply":"2023-06-21T08:11:04.695915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the best model with augmented data\nhistory = best_model.fit(epochs=2, validation_data=(x_test, y_test), callbacks=[es_callback])\n\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the best model on the test set\nscore = best_model.evaluate(x_test, y_test, verbose=0)\nprint('Test loss:', score[0])\nprint('Test accuracy:', score[1])\n","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:11:11.791949Z","iopub.execute_input":"2023-06-21T08:11:11.792597Z","iopub.status.idle":"2023-06-21T08:11:18.989122Z","shell.execute_reply.started":"2023-06-21T08:11:11.792558Z","shell.execute_reply":"2023-06-21T08:11:18.988023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_model.save('/kaggle/working/isic2017cnn@.h5')","metadata":{"execution":{"iopub.status.busy":"2023-06-05T08:17:18.259346Z","iopub.execute_input":"2023-06-05T08:17:18.259676Z","iopub.status.idle":"2023-06-05T08:17:19.446700Z","shell.execute_reply.started":"2023-06-05T08:17:18.259648Z","shell.execute_reply":"2023-06-05T08:17:19.445488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_prob = best_model.predict(x_test)\ny_test = y_test.argmax(axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:11:26.382078Z","iopub.execute_input":"2023-06-21T08:11:26.382952Z","iopub.status.idle":"2023-06-21T08:11:33.232644Z","shell.execute_reply.started":"2023-06-21T08:11:26.382900Z","shell.execute_reply":"2023-06-21T08:11:33.231458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_curve\n\nfpr, tpr, thresholds = roc_curve(y_test, y_pred_prob[:, 1])","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:11:35.848620Z","iopub.execute_input":"2023-06-21T08:11:35.848979Z","iopub.status.idle":"2023-06-21T08:11:35.855408Z","shell.execute_reply.started":"2023-06-21T08:11:35.848950Z","shell.execute_reply":"2023-06-21T08:11:35.854349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\n\nauc = roc_auc_score(y_test, y_pred_prob[:, 1])","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:11:37.926647Z","iopub.execute_input":"2023-06-21T08:11:37.927375Z","iopub.status.idle":"2023-06-21T08:11:37.934697Z","shell.execute_reply.started":"2023-06-21T08:11:37.927339Z","shell.execute_reply":"2023-06-21T08:11:37.933702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(fpr)):\n    print(\"Threshold: {:.2f} | FPR: {:.4f} | TPR: {:.4f}\".format(thresholds[i], fpr[i], tpr[i]))\n\nprint(\"AUC: {:.4f}\".format(auc))","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:11:39.686823Z","iopub.execute_input":"2023-06-21T08:11:39.687536Z","iopub.status.idle":"2023-06-21T08:11:39.695597Z","shell.execute_reply.started":"2023-06-21T08:11:39.687501Z","shell.execute_reply":"2023-06-21T08:11:39.694451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the ROC curve\nplt.plot(fpr, tpr, label='ROC curve (area = {:.4f})'.format(auc))\nplt.plot([0, 1], [0, 1], 'k--', label='Random guess')\nplt.xlabel('False positive rate')\nplt.ylabel('True positive rate')\nplt.title('ROC curve')\nplt.legend(loc='lower right')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:11:44.191914Z","iopub.execute_input":"2023-06-21T08:11:44.192768Z","iopub.status.idle":"2023-06-21T08:11:44.582680Z","shell.execute_reply.started":"2023-06-21T08:11:44.192735Z","shell.execute_reply":"2023-06-21T08:11:44.581745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = y_pred_prob.argmax(axis=1)\n\nfrom sklearn.metrics import confusion_matrix, classification_report\n\n# Assuming y_test and y_pred are the true and predicted labels, respectively\ncm = confusion_matrix(y_test, y_pred)\n\n# Print the confusion matrix\nprint(\"Confusion matrix:\")\nprint(cm)\n\n# Extract metrics from the confusion matrix\ntn, fp, fn, tp = cm.ravel()\n\n# Compute precision, recall, and F1-score\nprecision = tp / (tp + fp)\nrecall = tp / (tp + fn)\nf1_score = 2 * precision * recall / (precision + recall)\n\n# Print the metrics\nprint(\"Precision: {:.4f}\".format(precision))\nprint(\"Recall: {:.4f}\".format(recall))\nprint(\"F1-score: {:.4f}\".format(f1_score))\n\n# Generate a classification report\nprint(\"Classification report:\")\nprint(classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:11:48.688198Z","iopub.execute_input":"2023-06-21T08:11:48.689117Z","iopub.status.idle":"2023-06-21T08:11:48.707534Z","shell.execute_reply.started":"2023-06-21T08:11:48.689080Z","shell.execute_reply":"2023-06-21T08:11:48.706322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"precision = tp / (tp + fp)\nrecall = tp / (tp + fn)\nf1_score = 2 * precision * recall / (precision + recall)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:11:52.287429Z","iopub.execute_input":"2023-06-21T08:11:52.288178Z","iopub.status.idle":"2023-06-21T08:11:52.292861Z","shell.execute_reply.started":"2023-06-21T08:11:52.288145Z","shell.execute_reply":"2023-06-21T08:11:52.291915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom sklearn.metrics import confusion_matrix\nimport numpy as np\n\n# Assuming y_test and y_pred are the true and predicted labels, respectively\ncm = confusion_matrix(y_test, y_pred)\n\n# Define the names of the classes\nclasses = ['Class 0', 'Class 1']\n\n# Plot the confusion matrix\nplt.imshow(cm, interpolation='nearest', cmap=plt.cm.Blues)\nplt.colorbar()\nplt.title('Confusion Matrix')\nplt.xlabel('Predicted label')\nplt.ylabel('True label')\nplt.xticks(np.arange(len(classes)), classes)\nplt.yticks(np.arange(len(classes)), classes)\n\n# Add values to the plot\nthresh = cm.max() / 2.\nfor i, j in np.ndindex(cm.shape):\n    plt.text(j, i, format(cm[i, j], 'd'),\n             horizontalalignment=\"center\",\n             color=\"white\" if cm[i, j] > thresh else \"black\")\n\n# Save the image\nplt.savefig('confusion_matrix.png')\n\n# Show the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:11:54.359547Z","iopub.execute_input":"2023-06-21T08:11:54.360266Z","iopub.status.idle":"2023-06-21T08:11:54.675560Z","shell.execute_reply.started":"2023-06-21T08:11:54.360220Z","shell.execute_reply":"2023-06-21T08:11:54.674534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom sklearn.metrics import confusion_matrix, accuracy_score\nimport numpy as np\n\n# Assuming y_test and y_pred are the true and predicted labels, respectively\ncm = confusion_matrix(y_test, y_pred)\n\n# Calculate accuracy\naccuracy = accuracy_score(y_test, y_pred)\n\n# Calculate accuracy for each class\nclass_accuracy = cm.diagonal() / cm.sum(axis=1)\n\n# Plot bar chart of accuracy for each class\nfig, ax = plt.subplots()\nax.bar(np.arange(len(classes)), class_accuracy)\nax.set_xticks(np.arange(len(classes)))\nax.set_xticklabels(classes)\n\n# Add accuracy value as text on top of each bar\nfor i, v in enumerate(class_accuracy):\n    ax.text(i, v+0.01, f'{v:.2f}', ha='center')\n\n# Set plot title and axis labels\nax.set_title('Accuracy for Each Class')\nax.set_xlabel('Class')\nax.set_ylabel('Accuracy')\n\n# Show the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:11:58.155185Z","iopub.execute_input":"2023-06-21T08:11:58.155993Z","iopub.status.idle":"2023-06-21T08:11:58.340530Z","shell.execute_reply.started":"2023-06-21T08:11:58.155958Z","shell.execute_reply":"2023-06-21T08:11:58.339559Z"},"trusted":true},"execution_count":null,"outputs":[]}]}