{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# https://www.kaggle.com/code/jaykumar1607/brain-tumor-mri-classification-tensorflow-cnn/notebook\nfrom PIL import Image\nfrom PIL import ImageFilter\nfrom tensorflow.python.client import device_lib\nfrom tensorflow.keras.models import Sequential, Model\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau, TensorBoard, ModelCheckpoint\nfrom tensorflow.keras.losses import CategoricalCrossentropy\nfrom tensorflow import keras\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import EfficientNetB5\nfrom tensorflow.keras.applications.efficientnet import EfficientNetB3\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.models import Sequential, Model\nfrom tensorflow.keras.layers import Conv2D, Flatten, MaxPooling2D, Dense, Input, Reshape, Concatenate, GlobalAveragePooling2D, BatchNormalization, Dropout, Activation, GlobalMaxPooling2D\nfrom tensorflow.keras.utils import Sequence\nfrom tensorflow.compat.v1 import ConfigProto\nfrom tensorflow.compat.v1 import InteractiveSession\nfrom tensorflow.keras.optimizers import Adam\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, classification_report\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom skimage.io import imread\nfrom warnings import filterwarnings\nimport warnings\nwarnings.simplefilter(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2022-10-02T17:12:18.648245Z","iopub.execute_input":"2022-10-02T17:12:18.648614Z","iopub.status.idle":"2022-10-02T17:12:18.659972Z","shell.execute_reply.started":"2022-10-02T17:12:18.648582Z","shell.execute_reply":"2022-10-02T17:12:18.658924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport tensorflow as tk\nfrom keras import backend as K\nimport seaborn as sns\nimport os","metadata":{"execution":{"iopub.status.busy":"2022-10-02T17:12:18.662333Z","iopub.execute_input":"2022-10-02T17:12:18.662937Z","iopub.status.idle":"2022-10-02T17:12:18.671942Z","shell.execute_reply.started":"2022-10-02T17:12:18.6629Z","shell.execute_reply":"2022-10-02T17:12:18.670963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nDir = '../input/cassava-leaf-disease-classification'","metadata":{"execution":{"iopub.status.busy":"2022-10-02T17:12:18.673575Z","iopub.execute_input":"2022-10-02T17:12:18.673943Z","iopub.status.idle":"2022-10-02T17:12:18.683409Z","shell.execute_reply.started":"2022-10-02T17:12:18.673909Z","shell.execute_reply":"2022-10-02T17:12:18.682478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv('../input/cassava-leaf-disease-classification/train.csv')\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-02T17:12:18.684929Z","iopub.execute_input":"2022-10-02T17:12:18.685673Z","iopub.status.idle":"2022-10-02T17:12:18.712513Z","shell.execute_reply.started":"2022-10-02T17:12:18.685622Z","shell.execute_reply":"2022-10-02T17:12:18.71171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"file = open(os.path.join(path, 'label_num_to_disease_map.json'))\nclass_map = json.load(file)\nclass_map","metadata":{"execution":{"iopub.status.busy":"2022-09-21T03:05:29.477462Z","iopub.execute_input":"2022-09-21T03:05:29.47781Z","iopub.status.idle":"2022-09-21T03:05:29.720727Z","shell.execute_reply.started":"2022-09-21T03:05:29.477782Z","shell.execute_reply":"2022-09-21T03:05:29.718912Z"}}},{"cell_type":"code","source":"train_data['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-10-02T17:12:18.716071Z","iopub.execute_input":"2022-10-02T17:12:18.716919Z","iopub.status.idle":"2022-10-02T17:12:18.725028Z","shell.execute_reply.started":"2022-10-02T17:12:18.716885Z","shell.execute_reply":"2022-10-02T17:12:18.723917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_style(\"whitegrid\")\nfig, ax = plt.subplots(figsize = (6, 4))\n\nfor i in ['top', 'right', 'left']:\n    ax.spines[i].set_visible(False)\nax.spines['bottom'].set_color('black')\nsns.countplot(train_data['label'], edgecolor = 'black',\n              palette = reversed(sns.color_palette(\"viridis\", 5)))\nplt.xlabel('Classes', fontfamily = 'serif', size = 15)\nplt.ylabel('Count', fontfamily = 'serif', size = 15)\nplt.xticks(fontfamily = 'serif', size = 12)\nplt.yticks(fontfamily = 'serif', size = 12)\nax.grid(axis = 'y', linestyle = '--', alpha = 0.9)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-02T17:12:18.726695Z","iopub.execute_input":"2022-10-02T17:12:18.727181Z","iopub.status.idle":"2022-10-02T17:12:18.939324Z","shell.execute_reply.started":"2022-10-02T17:12:18.727145Z","shell.execute_reply":"2022-10-02T17:12:18.938238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"fig = plt.figure(figsize=(8, 6))\ntrain_data['label'] = train_data['label'].astype('str') \nsns.set(font_scale=1.1)\nlabel_count = sns.countplot(x='label', data=train_data, order = train_data['label'].value_counts().index)\nlabel_count.set_xticklabels(label_count.get_xticklabels(), rotation=10)\nplt.ylabel('Count', fontsize=12)\nplt.xlabel('Labels', fontsize=12)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-21T03:05:29.724921Z","iopub.status.idle":"2022-09-21T03:05:29.725433Z","shell.execute_reply.started":"2022-09-21T03:05:29.725162Z","shell.execute_reply":"2022-09-21T03:05:29.725184Z"}}},{"cell_type":"markdown","source":"sample = train_labels[train_labels.label == 0].sample(1)\nplt.figure(figsize=(15, 5))\nfor ind, (image_id, label) in enumerate(zip(sample.image_id, sample.label)):\n    plt.subplot(1, 5, ind + 1)\n    img = cv2.imread(os.path.join(WORK_DIR, \"train_images\", image_id))\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    plt.imshow(img)\n    plt.axis(\"off\")\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:52:21.068025Z","iopub.execute_input":"2022-09-21T05:52:21.068775Z","iopub.status.idle":"2022-09-21T05:52:21.355479Z","shell.execute_reply.started":"2022-09-21T05:52:21.068741Z","shell.execute_reply":"2022-09-21T05:52:21.353309Z"}}},{"cell_type":"code","source":"batch_size = 8\nSTEPS_PER_EPOCH = len(train_data['label']) * 0.8 / batch_size\nVALIDATION_STEPS = len(train_data['label']) * 0.2 / batch_size\ntarget_size_dim = 512 ","metadata":{"execution":{"iopub.status.busy":"2022-10-02T17:12:18.940882Z","iopub.execute_input":"2022-10-02T17:12:18.941313Z","iopub.status.idle":"2022-10-02T17:12:18.946629Z","shell.execute_reply.started":"2022-10-02T17:12:18.941277Z","shell.execute_reply":"2022-10-02T17:12:18.945715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['label'] = train_data['label'].astype('string')\n\ntrain_datagen = ImageDataGenerator(validation_split = 0.2,\n                                   preprocessing_function = None,\n                                   rotation_range = 45,\n                                   zoom_range = 0.2,\n                                   horizontal_flip = True,\n                                   vertical_flip = True,\n                                   fill_mode = 'nearest',\n                                   shear_range = 0.1,\n                                   height_shift_range = 0.1,\n                                   width_shift_range = 0.1)\n\ntrain_generator = train_datagen.flow_from_dataframe(train_data,\n                                                    directory = os.path.join(Dir, \"train_images\"),\n                                                    subset = \"training\",\n                                                    x_col = \"image_id\",\n                                                    y_col = \"label\",\n                                                    target_size = (target_size_dim,target_size_dim),\n                                                    batch_size = batch_size,\n                                                    class_mode = \"sparse\",\n                                                    seed = 8)\n\nvalidation_datagen = ImageDataGenerator(validation_split = 0.2)\n\nvalidation_generator = validation_datagen.flow_from_dataframe(train_data,\n                                                              directory = os.path.join(Dir, \"train_images\"),\n                                                              subset = \"validation\",\n                                                              x_col = \"image_id\",\n                                                              y_col = \"label\",\n                                                              target_size = (target_size_dim,target_size_dim),\n                                                              batch_size = batch_size,\n                                                              class_mode = \"sparse\",\n                                                              seed = 8)\n","metadata":{"execution":{"iopub.status.busy":"2022-10-02T17:12:18.948219Z","iopub.execute_input":"2022-10-02T17:12:18.948939Z","iopub.status.idle":"2022-10-02T17:12:43.532743Z","shell.execute_reply.started":"2022-10-02T17:12:18.948904Z","shell.execute_reply":"2022-10-02T17:12:43.531702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"efficientnetb3 = EfficientNetB3(weights=None,\n                                input_shape=(target_size_dim,target_size_dim,3),\n                                include_top=False)\nmodel = efficientnetb3.output\nmodel = tk.keras.layers.GlobalAveragePooling2D()(model)\nmodel = tk.keras.layers.Dropout(rate=0.4)(model)\nmodel = tk.keras.layers.Dense(5,activation='softmax')(model)\nmodel = tk.keras.models.Model(inputs=efficientnetb3.input, outputs = model)\nmodel.summary()","metadata":{}},{"cell_type":"code","source":"model = Sequential()\nmodel.add(EfficientNetB0(weights=None, \n                         include_top=False, \n                         input_shape=(target_size_dim, target_size_dim, 3)))\n#weight = '../input/efficientnetb3-notop/efficientnetb3_notop.h5'\n    \nmodel.add(GlobalAveragePooling2D())\n#model.add(Dense(256))\n#model.add(BatchNormalization())\n#model.add(Activation('relu'))\n#model.add(Dropout(0.3))\nmodel.add(Dense(5, activation='softmax'))\nmodel.summary()","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-10-02T17:12:43.53598Z","iopub.execute_input":"2022-10-02T17:12:43.536344Z","iopub.status.idle":"2022-10-02T17:12:45.533595Z","shell.execute_reply.started":"2022-10-02T17:12:43.536316Z","shell.execute_reply":"2022-10-02T17:12:45.532575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#predictions = model(train_data[0:5]).numpy()\n#predictions","metadata":{"execution":{"iopub.status.busy":"2022-10-02T17:12:45.535261Z","iopub.execute_input":"2022-10-02T17:12:45.535825Z","iopub.status.idle":"2022-10-02T17:12:45.54133Z","shell.execute_reply.started":"2022-10-02T17:12:45.535788Z","shell.execute_reply":"2022-10-02T17:12:45.540336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"def weighted_loss(labels,logits):\n    sf_logits_log = (-1) * K.log(logits)\n    num_class = logits.shape[-1]\n    oh_labels = K.one_hot(labels, num_class, dtype = tf.float32)\n    y_true = 1.2 * oh_labels[:, 1:]\n    y_false = 0.8 * oh_labels[:, 0:1]\n    weight_labels = K.concatenate([y_false, y_true], axis = 1)\n    lodd = K.sum(sf_logits_log * weight_labels, axis = 1)\n    loss = K.mean(loss)\n    return loss","metadata":{"execution":{"iopub.status.busy":"2022-09-28T07:46:45.450504Z","iopub.execute_input":"2022-09-28T07:46:45.451096Z","iopub.status.idle":"2022-09-28T07:46:45.459348Z","shell.execute_reply.started":"2022-09-28T07:46:45.451058Z","shell.execute_reply":"2022-09-28T07:46:45.458094Z"}}},{"cell_type":"code","source":"#labels = tf.constant([1, 0, 1, 0, 0, 0, 0, 0, 1])\n#logits = tf.constant([1.0, -2.0, 7.9, 1.1, 7.1, -13.2, -13.1, 3.09, 6.3])\nmodel.compile(loss = \"sparse_categorical_crossentropy\",\n              optimizer = Adam(lr = 0.001), \n              metrics= ['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-10-02T17:12:45.542846Z","iopub.execute_input":"2022-10-02T17:12:45.54391Z","iopub.status.idle":"2022-10-02T17:12:45.558063Z","shell.execute_reply.started":"2022-10-02T17:12:45.543875Z","shell.execute_reply":"2022-10-02T17:12:45.557097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#tensorboard = TensorBoard(log_dir = 'logs')\nearly_stopping = EarlyStopping(monitor='val_loss', \n                               min_delta = 0.001,\n                               mode='min', \n                               patience=5, \n                               restore_best_weights=True,\n                               verbose=1)\ncheckpoint = ModelCheckpoint(\"EffNetB0_512_8_best_weights.h5\",\n                             monitor=\"val_loss\",\n                             save_best_only=True,\n                             save_weights_only = True,\n                             mode=\"min\",\n                             verbose=1)\nreduce_lr = ReduceLROnPlateau(monitor = 'val_loss', \n                              factor = 0.3, \n                              patience = 2, \n                              min_lr=0.001,\n                              mode='min',\n                              verbose=1 )","metadata":{"execution":{"iopub.status.busy":"2022-10-02T17:12:45.559394Z","iopub.execute_input":"2022-10-02T17:12:45.559734Z","iopub.status.idle":"2022-10-02T17:12:45.567967Z","shell.execute_reply.started":"2022-10-02T17:12:45.559697Z","shell.execute_reply":"2022-10-02T17:12:45.566849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train_generator,\n                    steps_per_epoch = STEPS_PER_EPOCH,\n                    validation_data = validation_generator,\n                    validation_steps = VALIDATION_STEPS,\n                    epochs = 20, \n                    verbose=1, \n                    batch_size=16,\n                    callbacks=[early_stopping,checkpoint,reduce_lr])","metadata":{"execution":{"iopub.status.busy":"2022-10-02T17:12:45.569601Z","iopub.execute_input":"2022-10-02T17:12:45.569969Z","iopub.status.idle":"2022-10-03T00:24:05.386797Z","shell.execute_reply.started":"2022-10-02T17:12:45.56993Z","shell.execute_reply":"2022-10-03T00:24:05.385845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filterwarnings('ignore')\n\nepochs = [i for i in range(20)]\nfig, ax = plt.subplots(1,2,figsize=(14,7))\ntrain_acc = history.history['accuracy']\ntrain_loss = history.history['loss']\nval_acc = history.history['val_accuracy']\nval_loss = history.history['val_loss']\n\nfig.text(s='Epochs vs. Training and Validation Accuracy/Loss',\n         size=18,\n         fontweight='bold',\n         fontname='monospace',\n         y=1,\n         x=0.28,\n         alpha=0.8)\n\nsns.despine()\nax[0].plot(epochs, \n           train_acc,\n           marker='o',\n           label ='Training Accuracy')\nax[0].plot(epochs, \n           val_acc, \n           marker='o',\n           label = 'Validation Accuracy')\nax[0].legend(frameon=False)\nax[0].set_xlabel('Epochs')\nax[0].set_ylabel('Accuracy')\n\nsns.despine()\nax[1].plot(epochs, \n           train_loss, \n           marker='o',\n           label ='Training Loss')\nax[1].plot(epochs, \n           val_loss, \n           marker='o',\n           label = 'Validation Loss')\nax[1].legend(frameon=False)\nax[1].set_xlabel('Epochs')\nax[1].set_ylabel('Training & Validation Loss')\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-03T00:24:40.906537Z","iopub.execute_input":"2022-10-03T00:24:40.906921Z","iopub.status.idle":"2022-10-03T00:24:41.393607Z","shell.execute_reply.started":"2022-10-03T00:24:40.906891Z","shell.execute_reply":"2022-10-03T00:24:41.392731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_valid_y = model.predict(validation_generator,  verbose = True)\npred_valid_y_labels = np.argmax(pred_valid_y, axis=-1)\nvalid_labels=validation_generator.labels\nprint(classification_report(valid_labels, pred_valid_y_labels ))\n","metadata":{"execution":{"iopub.status.busy":"2022-10-03T00:24:46.816805Z","iopub.execute_input":"2022-10-03T00:24:46.817206Z","iopub.status.idle":"2022-10-03T00:25:40.473668Z","shell.execute_reply.started":"2022-10-03T00:24:46.817173Z","shell.execute_reply":"2022-10-03T00:25:40.472614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm = confusion_matrix(valid_labels, pred_valid_y_labels)#, normalize='true'\nsns.heatmap(cm, annot=True, fmt=\"d\")","metadata":{"execution":{"iopub.status.busy":"2022-10-03T00:25:40.47558Z","iopub.execute_input":"2022-10-03T00:25:40.475964Z","iopub.status.idle":"2022-10-03T00:25:40.786444Z","shell.execute_reply.started":"2022-10-03T00:25:40.475926Z","shell.execute_reply":"2022-10-03T00:25:40.785547Z"},"trusted":true},"execution_count":null,"outputs":[]}]}