{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-05T08:20:21.260032Z","iopub.execute_input":"2023-02-05T08:20:21.260608Z","iopub.status.idle":"2023-02-05T08:20:23.139663Z","shell.execute_reply.started":"2023-02-05T08:20:21.260523Z","shell.execute_reply":"2023-02-05T08:20:23.138583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom pathlib import Path\nimport os.path\nimport matplotlib.pyplot as plt\nfrom IPython.display import Image, display, Markdown\nimport matplotlib.cm as cm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix\nimport tensorflow as tf\nfrom time import perf_counter\nimport seaborn as sns\n\ndef printmd(string):\n    # Print with Markdowns    \n    display(Markdown(string))","metadata":{"execution":{"iopub.status.busy":"2023-02-05T08:20:23.141950Z","iopub.execute_input":"2023-02-05T08:20:23.142346Z","iopub.status.idle":"2023-02-05T08:20:29.006191Z","shell.execute_reply.started":"2023-02-05T08:20:23.142309Z","shell.execute_reply":"2023-02-05T08:20:29.005159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_dir = Path('/kaggle/input/diabetic-retinopathy-224x224-gaussian-filtered/gaussian_filtered_images')\n\n# Get filepaths and labels\nfilepaths = list(image_dir.glob(r'**/*.png'))\nlabels = list(map(lambda x: os.path.split(os.path.split(x)[0])[1], filepaths))","metadata":{"execution":{"iopub.status.busy":"2023-02-05T08:20:29.007515Z","iopub.execute_input":"2023-02-05T08:20:29.009046Z","iopub.status.idle":"2023-02-05T08:20:29.449216Z","shell.execute_reply.started":"2023-02-05T08:20:29.009006Z","shell.execute_reply":"2023-02-05T08:20:29.448215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filepaths = pd.Series(filepaths, name='Filepath').astype(str)\nlabels = pd.Series(labels, name='Label')\n\n# Concatenate filepaths and labels\nimage_df = pd.concat([filepaths, labels], axis=1)\n\n# Shuffle the DataFrame and reset index\nimage_df = image_df.sample(frac=1).reset_index(drop = True)\n\n# Show the result\nimage_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T08:20:29.453605Z","iopub.execute_input":"2023-02-05T08:20:29.453939Z","iopub.status.idle":"2023-02-05T08:20:29.486200Z","shell.execute_reply.started":"2023-02-05T08:20:29.453909Z","shell.execute_reply":"2023-02-05T08:20:29.485201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display some pictures of the dataset with their labels\nfig, axes = plt.subplots(nrows=3, ncols=4, figsize=(10, 7),\n                        subplot_kw={'xticks': [], 'yticks': []})\n\nfor i, ax in enumerate(axes.flat):\n    ax.imshow(plt.imread(image_df.Filepath[i]))\n    ax.set_title(image_df.Label[i])\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-05T08:20:29.487439Z","iopub.execute_input":"2023-02-05T08:20:29.487775Z","iopub.status.idle":"2023-02-05T08:20:30.530529Z","shell.execute_reply.started":"2023-02-05T08:20:29.487742Z","shell.execute_reply":"2023-02-05T08:20:30.529689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_gen():\n    # Load the Images with a generator and Data Augmentation\n    train_generator = tf.keras.preprocessing.image.ImageDataGenerator(\n        preprocessing_function=tf.keras.applications.mobilenet_v2.preprocess_input,\n        validation_split=0.1\n    )\n\n    test_generator = tf.keras.preprocessing.image.ImageDataGenerator(\n        preprocessing_function=tf.keras.applications.mobilenet_v2.preprocess_input\n    )\n\n    train_images = train_generator.flow_from_dataframe(\n        dataframe=train_df,\n        x_col='Filepath',\n        y_col='Label',\n        target_size=(224, 224),\n        color_mode='rgb',\n        class_mode='categorical',\n        batch_size=32,\n        shuffle=True,\n        seed=0,\n        subset='training',\n        rotation_range=30, # Uncomment to use data augmentation\n        zoom_range=0.15,\n        width_shift_range=0.2,\n        height_shift_range=0.2,\n        shear_range=0.15,\n        horizontal_flip=True,\n        fill_mode=\"nearest\"\n    )\n\n    val_images = train_generator.flow_from_dataframe(\n        dataframe=train_df,\n        x_col='Filepath',\n        y_col='Label',\n        target_size=(224, 224),\n        color_mode='rgb',\n        class_mode='categorical',\n        batch_size=32,\n        shuffle=True,\n        seed=0,\n        subset='validation',\n        rotation_range=30, # Uncomment to use data augmentation\n        zoom_range=0.15,\n        width_shift_range=0.2,\n        height_shift_range=0.2,\n        shear_range=0.15,\n        horizontal_flip=True,\n        fill_mode=\"nearest\"\n    )\n\n    test_images = test_generator.flow_from_dataframe(\n        dataframe=test_df,\n        x_col='Filepath',\n        y_col='Label',\n        target_size=(224, 224),\n        color_mode='rgb',\n        class_mode='categorical',\n        batch_size=32,\n        shuffle=False\n    )\n    \n    return train_generator,test_generator,train_images,val_images,test_images","metadata":{"execution":{"iopub.status.busy":"2023-02-05T08:20:30.532721Z","iopub.execute_input":"2023-02-05T08:20:30.533065Z","iopub.status.idle":"2023-02-05T08:20:30.546070Z","shell.execute_reply.started":"2023-02-05T08:20:30.533033Z","shell.execute_reply":"2023-02-05T08:20:30.544704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Map the labels to have only \"No_DR\" and \"DR\"\nimage_df_red = image_df.copy()\nimage_df_red['Label'] = image_df_red['Label'].apply(lambda x: x if x == 'No_DR' else 'DR')\nimage_df_red.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T08:20:30.547829Z","iopub.execute_input":"2023-02-05T08:20:30.548517Z","iopub.status.idle":"2023-02-05T08:20:30.569088Z","shell.execute_reply.started":"2023-02-05T08:20:30.548475Z","shell.execute_reply":"2023-02-05T08:20:30.567955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display the number of pictures of each category\nvc = image_df_red['Label'].value_counts()\nplt.figure(figsize=(9,5))\nsns.barplot(x = vc.index, y = vc, palette = \"rocket\")\nplt.title(\"Number of pictures of each category\", fontsize = 15)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-05T08:20:30.570984Z","iopub.execute_input":"2023-02-05T08:20:30.571548Z","iopub.status.idle":"2023-02-05T08:20:30.765183Z","shell.execute_reply.started":"2023-02-05T08:20:30.571512Z","shell.execute_reply":"2023-02-05T08:20:30.764133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Separate in train and test data\ntrain_df, test_df = train_test_split(image_df_red, train_size=0.9, shuffle=True, random_state=1)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T08:20:30.768269Z","iopub.execute_input":"2023-02-05T08:20:30.768544Z","iopub.status.idle":"2023-02-05T08:20:30.775134Z","shell.execute_reply.started":"2023-02-05T08:20:30.768517Z","shell.execute_reply":"2023-02-05T08:20:30.774098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create the generators\ntrain_generator,test_generator,train_images,val_images,test_images=create_gen()\n\n# Load the pretained model\npretrained_model = tf.keras.applications.MobileNetV2(\n    input_shape=(224, 224, 3),\n    include_top=False,\n    weights='imagenet',\n    pooling='avg'\n)\n\npretrained_model.trainable = False\n\n\ninputs = pretrained_model.input\n\nx = tf.keras.layers.Dense(128, activation='relu')(pretrained_model.output)\nx = tf.keras.layers.Dense(128, activation='relu')(x)\n\noutputs = tf.keras.layers.Dense(2, activation='softmax')(x)\n\nmodel = tf.keras.Model(inputs=inputs, outputs=outputs)\n\nmodel.compile(\n    optimizer='adam',\n    loss='categorical_crossentropy',\n    metrics=['accuracy']\n)\n\nhistory = model.fit(\n    train_images,\n    validation_data=val_images,\n    batch_size = 32,\n    epochs=20,\n    callbacks=[\n        tf.keras.callbacks.EarlyStopping(\n            monitor='val_loss',\n            patience=3,\n            restore_best_weights=True\n        )\n    ]\n)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T08:28:28.413889Z","iopub.execute_input":"2023-02-05T08:28:28.414858Z","iopub.status.idle":"2023-02-05T08:30:03.565638Z","shell.execute_reply.started":"2023-02-05T08:28:28.414819Z","shell.execute_reply":"2023-02-05T08:30:03.564461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(history.history)[['accuracy','val_accuracy']].plot()\nplt.title(\"Accuracy\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-05T08:31:17.393385Z","iopub.execute_input":"2023-02-05T08:31:17.394201Z","iopub.status.idle":"2023-02-05T08:31:17.977263Z","shell.execute_reply.started":"2023-02-05T08:31:17.394146Z","shell.execute_reply":"2023-02-05T08:31:17.975158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(history.history)[['loss','val_loss']].plot()\nplt.title(\"Loss\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-05T08:31:31.244226Z","iopub.execute_input":"2023-02-05T08:31:31.244940Z","iopub.status.idle":"2023-02-05T08:31:31.468331Z","shell.execute_reply.started":"2023-02-05T08:31:31.244904Z","shell.execute_reply":"2023-02-05T08:31:31.467419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = model.evaluate(test_images, verbose=0)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T08:31:44.188480Z","iopub.execute_input":"2023-02-05T08:31:44.189378Z","iopub.status.idle":"2023-02-05T08:31:45.593984Z","shell.execute_reply.started":"2023-02-05T08:31:44.189343Z","shell.execute_reply":"2023-02-05T08:31:45.592878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"printmd(\" ## Test Loss: {:.5f}\".format(results[0]))\nprintmd(\"## Accuracy on the test set: {:.2f}%\".format(results[1] * 100))\nprint('\\n')","metadata":{"execution":{"iopub.status.busy":"2023-02-05T08:31:47.535422Z","iopub.execute_input":"2023-02-05T08:31:47.535809Z","iopub.status.idle":"2023-02-05T08:31:47.546673Z","shell.execute_reply.started":"2023-02-05T08:31:47.535775Z","shell.execute_reply":"2023-02-05T08:31:47.545499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict the label of the test_images\npred = model.predict(test_images)\npred = np.argmax(pred,axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-05T08:32:01.450996Z","iopub.execute_input":"2023-02-05T08:32:01.451396Z","iopub.status.idle":"2023-02-05T08:32:03.466198Z","shell.execute_reply.started":"2023-02-05T08:32:01.451361Z","shell.execute_reply":"2023-02-05T08:32:03.465064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Map the label\nlabels = (train_images.class_indices)\nlabels = dict((v,k) for k,v in labels.items())\npred = [labels[k] for k in pred]\n\n# Display the result\nprint(f'The first 5 predictions: {pred[:5]}')\n\nfrom sklearn.metrics import classification_report\ny_test = list(test_df.Label)\nprint(classification_report(y_test, pred))\n\ncf_matrix = confusion_matrix(y_test, pred, normalize='true')\nplt.figure(figsize = (10,6))\nsns.heatmap(cf_matrix, annot=True, xticklabels = sorted(set(y_test)), yticklabels = sorted(set(y_test)))\nplt.title('Normalized Confusion Matrix')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-05T08:32:10.805919Z","iopub.execute_input":"2023-02-05T08:32:10.806436Z","iopub.status.idle":"2023-02-05T08:32:11.110096Z","shell.execute_reply.started":"2023-02-05T08:32:10.806387Z","shell.execute_reply":"2023-02-05T08:32:11.109155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras import layers\nCNN_Model = tf.keras.Sequential([\n    layers.Conv2D(8, (3,3), padding=\"valid\", input_shape=(224,224,3), activation = 'relu'),\n    layers.MaxPooling2D(pool_size=(2,2)),\n    layers.BatchNormalization(),\n    \n    layers.Conv2D(16, (3,3), padding=\"valid\", activation = 'relu'),\n    layers.MaxPooling2D(pool_size=(2,2)),\n    layers.BatchNormalization(),\n    \n    layers.Conv2D(32, (4,4), padding=\"valid\", activation = 'relu'),\n    layers.MaxPooling2D(pool_size=(2,2)),\n    layers.BatchNormalization(),\n \n    layers.Flatten(),\n    layers.Dense(32, activation = 'relu'),\n    layers.Dropout(0.2),\n    layers.Dense(2, activation = 'softmax')\n])\n\nCNN_Model.compile(optimizer=tf.keras.optimizers.Adam(lr = 1e-5),\n              loss=tf.keras.losses.BinaryCrossentropy(),\n              metrics=['acc'])\n\nhistory = CNN_Model.fit(train_images,\n                    epochs=15,\n                    validation_data=val_images)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T08:47:48.699807Z","iopub.execute_input":"2023-02-05T08:47:48.700239Z","iopub.status.idle":"2023-02-05T08:50:51.292096Z","shell.execute_reply.started":"2023-02-05T08:47:48.700203Z","shell.execute_reply":"2023-02-05T08:50:51.291033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CNN_results = CNN_Model.evaluate(test_images, verbose=0)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T08:51:15.731992Z","iopub.execute_input":"2023-02-05T08:51:15.732398Z","iopub.status.idle":"2023-02-05T08:51:17.376258Z","shell.execute_reply.started":"2023-02-05T08:51:15.732364Z","shell.execute_reply":"2023-02-05T08:51:17.375237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"printmd(\" ## Test Loss: {:.5f}\".format(CNN_results[0]))\nprintmd(\"## Accuracy on the test set: {:.2f}%\".format(CNN_results[1] * 100))\nprint('\\n')","metadata":{"execution":{"iopub.status.busy":"2023-02-05T08:51:45.740368Z","iopub.execute_input":"2023-02-05T08:51:45.740803Z","iopub.status.idle":"2023-02-05T08:51:45.753434Z","shell.execute_reply.started":"2023-02-05T08:51:45.740772Z","shell.execute_reply":"2023-02-05T08:51:45.752030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict the label of the test_images\ncnn_pred = model.predict(test_images)\ncnn_pred = np.argmax(cnn_pred,axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T08:53:18.300589Z","iopub.execute_input":"2023-02-05T08:53:18.301484Z","iopub.status.idle":"2023-02-05T08:53:20.271720Z","shell.execute_reply.started":"2023-02-05T08:53:18.301448Z","shell.execute_reply":"2023-02-05T08:53:20.270542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Map the label\nlabels = (train_images.class_indices)\nlabels = dict((v,k) for k,v in labels.items())\ncnn_pred = [labels[k] for k in cnn_pred]\n\n# Display the result\nprint(f'The first 5 predictions: {cnn_pred[:5]}')\n\nfrom sklearn.metrics import classification_report\ny_test = list(test_df.Label)\nprint(classification_report(y_test, cnn_pred))\n\ncf_matrix = confusion_matrix(y_test, cnn_pred, normalize='true')\nplt.figure(figsize = (10,6))\nsns.heatmap(cf_matrix, annot=True, xticklabels = sorted(set(y_test)), yticklabels = sorted(set(y_test)))\nplt.title('Normalized Confusion Matrix')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-05T08:54:37.302665Z","iopub.execute_input":"2023-02-05T08:54:37.303066Z","iopub.status.idle":"2023-02-05T08:54:37.546129Z","shell.execute_reply.started":"2023-02-05T08:54:37.303033Z","shell.execute_reply":"2023-02-05T08:54:37.545203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('MobileNetV2.h5')","metadata":{"execution":{"iopub.status.busy":"2023-02-05T08:57:17.914704Z","iopub.execute_input":"2023-02-05T08:57:17.915073Z","iopub.status.idle":"2023-02-05T08:57:18.206684Z","shell.execute_reply.started":"2023-02-05T08:57:17.915041Z","shell.execute_reply":"2023-02-05T08:57:18.205665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CNN_Model.save('CNN Model.h5')","metadata":{"execution":{"iopub.status.busy":"2023-02-05T08:57:50.994371Z","iopub.execute_input":"2023-02-05T08:57:50.994751Z","iopub.status.idle":"2023-02-05T08:57:51.057654Z","shell.execute_reply.started":"2023-02-05T08:57:50.994719Z","shell.execute_reply":"2023-02-05T08:57:51.056659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\n\n\ndef predict_class(path):\n    img = cv2.imread(path)\n\n    RGBImg = cv2.cvtColor(img,cv2.COLOR_BGR2RGB)\n    RGBImg= cv2.resize(RGBImg,(224,224))\n    plt.imshow(RGBImg)\n    plt.xticks([])\n    plt.yticks([])\n    image = np.array(RGBImg) / 255.0\n    new_model = tf.keras.models.load_model(\"CNN Model.h5\")\n    predict=new_model.predict(np.array([image]))\n    per=np.argmax(predict,axis=1)\n    if per==1:\n        print('No DR')\n    else:\n        print('DR')\n    ","metadata":{"execution":{"iopub.status.busy":"2023-02-05T09:02:48.592495Z","iopub.execute_input":"2023-02-05T09:02:48.592931Z","iopub.status.idle":"2023-02-05T09:02:48.601483Z","shell.execute_reply.started":"2023-02-05T09:02:48.592895Z","shell.execute_reply":"2023-02-05T09:02:48.600200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict_class('/kaggle/input/diabetic-retina-images/img6.jpg')","metadata":{"execution":{"iopub.status.busy":"2023-02-05T09:03:31.772056Z","iopub.execute_input":"2023-02-05T09:03:31.772764Z","iopub.status.idle":"2023-02-05T09:03:32.187494Z","shell.execute_reply.started":"2023-02-05T09:03:31.772728Z","shell.execute_reply":"2023-02-05T09:03:32.185898Z"},"trusted":true},"execution_count":null,"outputs":[]}]}