{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"---------------------------------","metadata":{}},{"cell_type":"markdown","source":"# **II. EXPLORATORY DATA ANALYSIS**","metadata":{}},{"cell_type":"markdown","source":"## **1. Importing Libraries**","metadata":{}},{"cell_type":"code","source":"#Basic Libraries\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm,tqdm_notebook\nfrom prettytable import PrettyTable\nimport pickle\nimport os\nprint('CWD is ',os.getcwd())\n\n#Visualization Libraries\nfrom sklearn.manifold import TSNE\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nplt.rcParams[\"axes.grid\"] = False\n\n#Image Libraries\nfrom PIL import Image\nimport cv2\n\n#DL Libraries\nimport keras\nfrom keras import applications\nfrom tensorflow.keras.utils import img_to_array,array_to_img,load_img\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras import optimizers,Model,Sequential\nfrom keras.layers import Input,GlobalAveragePooling2D,Dropout,Dense,Activation\nfrom keras.callbacks import EarlyStopping,ReduceLROnPlateau","metadata":{"execution":{"iopub.status.busy":"2024-02-04T16:24:00.843114Z","iopub.execute_input":"2024-02-04T16:24:00.843478Z","iopub.status.idle":"2024-02-04T16:24:17.395335Z","shell.execute_reply.started":"2024-02-04T16:24:00.843445Z","shell.execute_reply":"2024-02-04T16:24:17.394382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **2. Visualization of Images**","metadata":{}},{"cell_type":"code","source":"'''This function reads data from the respective train and test directories'''\n\ndef load_data():\n    train = pd.read_csv('/kaggle/input/aptos2019-blindness-detection/train.csv')\n    test = pd.read_csv('/kaggle/input/aptos2019-blindness-detection/test.csv')\n    \n    train_dir = '/kaggle/input/aptos2019-blindness-detection/train_images'\n    test_dir = '/kaggle/input/aptos2019-blindness-detection/test_images'\n\n    \n    train['file_path'] = train['id_code'].map(lambda x: os.path.join(train_dir,'{}.png'.format(x)))\n    test['file_path'] = test['id_code'].map(lambda x: os.path.join(test_dir,'{}.png'.format(x)))\n    \n    train['file_name'] = train[\"id_code\"].apply(lambda x: x + \".png\")\n    test['file_name'] = test[\"id_code\"].apply(lambda x: x + \".png\")\n    \n    train['diagnosis'] = train['diagnosis'].astype(str)\n    \n    return train, test","metadata":{"execution":{"iopub.status.busy":"2024-02-04T16:24:17.397122Z","iopub.execute_input":"2024-02-04T16:24:17.39827Z","iopub.status.idle":"2024-02-04T16:24:17.405768Z","shell.execute_reply.started":"2024-02-04T16:24:17.398233Z","shell.execute_reply":"2024-02-04T16:24:17.404735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train,df_test = load_data()\nprint(df_train.shape,df_test.shape,'\\n')\ndf_train.head(6)","metadata":{"execution":{"iopub.status.busy":"2024-02-04T16:24:17.406982Z","iopub.execute_input":"2024-02-04T16:24:17.407232Z","iopub.status.idle":"2024-02-04T16:24:17.472469Z","shell.execute_reply.started":"2024-02-04T16:24:17.407211Z","shell.execute_reply":"2024-02-04T16:24:17.47155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\n# Specify the directory where you want to create folders\nbase_directory = '/kaggle/working/preprocessed_images/'\n\n# Create folders for each diagnosis grade\nfor grade in df_train['diagnosis'].unique():\n    folder_path = os.path.join(base_directory, f'grade_{grade}')\n    os.makedirs(folder_path, exist_ok=True)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-04T16:24:17.474495Z","iopub.execute_input":"2024-02-04T16:24:17.474935Z","iopub.status.idle":"2024-02-04T16:24:17.483945Z","shell.execute_reply.started":"2024-02-04T16:24:17.474902Z","shell.execute_reply":"2024-02-04T16:24:17.483065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# '''This Function performs image processing on top of images by performing Gaussian Blur and Circle Crop'''\n\n# def crop_image_from_gray(img,tol=7):\n#     if img.ndim ==2:\n#         mask = img>tol\n#         return img[np.ix_(mask.any(1),mask.any(0))]\n#     elif img.ndim==3:\n#         gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n#         mask = gray_img>tol\n        \n#         check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n#         if (check_shape == 0): # image is too dark so that we crop out everything,\n#             return img # return original image\n#         else:\n#             img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n#             img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n#             img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n#     #         print(img1.shape,img2.shape,img3.shape)\n#             img = np.stack([img1,img2,img3],axis=-1)\n#     #         print(img.shape)\n#         return img\n    \n    \n# def circle_crop(img, sigmaX):   \n#     \"\"\"Create circular crop around image centre\"\"\"    \n#     img = crop_image_from_gray(img)    \n#     img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        \n#     height, width, depth = img.shape    \n    \n#     x = int(width/2)\n#     y = int(height/2)\n#     r = np.amin((x,y))\n    \n#     circle_img = np.zeros((height, width), np.uint8)\n#     cv2.circle(circle_img, (x,y), int(r), 1, thickness=-1)\n#     img = cv2.bitwise_and(img, img, mask=circle_img)\n#     img = crop_image_from_gray(img)\n#     img=cv2.addWeighted(img,4, cv2.GaussianBlur( img , (0,0) , sigmaX) ,-4 ,128)\n#     return img \n\n\n\n# def extract_green_channel(image_path):\n#     # Read the image\n#     img = cv2.imread(image_path)\n\n#     # Extract the green channel\n#     green_channel = img[:, :, 1]\n\n#     return green_channel\n\n\n# def apply_preprocessing(img_path, sigmaX=50, clip_limit=5.0, tile_grid_size=(8, 8), target_size=(224, 224)):\n#     \"\"\"\n#     Apply the same preprocessing steps as in visualize_img_clahe to an image.\n\n#     Parameters:\n#     - img_path: Path to the original image.\n#     - sigmaX: Parameter for Gaussian blur.\n#     - clip_limit: Threshold for contrast limiting in CLAHE.\n#     - tile_grid_size: Size of the grid for histogram equalization in CLAHE.\n#     - target_size: Size to which the image should be resized.\n\n#     Returns:\n#     - img_preprocessed: Preprocessed and resized image.\n#     \"\"\"\n#     img = cv2.imread(img_path)\n# #     print(img)\n#     img = circle_crop(img, sigmaX)\n# #     green_channel = extract_green_channel(img)\n\n#     img_gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n\n# #     # Apply Gaussian blur\n# #     img_blur = cv2.addWeighted(img_gray, 4, cv2.GaussianBlur(img_gray, (0, 0), sigmaX), -4, 128)\n# #     img_gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n# #     img_gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n# #     _, img_thresh = cv2.threshold(img_gray, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)\n\n#     # Resize the image to the target size\n# #     img_resized = cv2.resize(img, target_size)\n\n#     # Apply CLAHE to the resized image if needed\n# #     clahe = cv2.createCLAHE(clipLimit=clip_limit, tileGridSize=tile_grid_size)\n# #     img_clahe = clahe.apply(img_gray)\n#     img_blur = cv2.addWeighted(img_gray, 4, cv2.GaussianBlur(img_gray, (0,0), sigmaX), -4 , 128)\n#     img_smooth = cv2.bilateralFilter(img_blur, 9, 75, 75)\n\n#     # Uncomment the line above if CLAHE is needed, or return img_resized directly\n#     return img_smooth\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-02-04T10:06:38.779037Z","iopub.execute_input":"2024-02-04T10:06:38.779426Z","iopub.status.idle":"2024-02-04T10:06:38.796902Z","shell.execute_reply.started":"2024-02-04T10:06:38.779391Z","shell.execute_reply":"2024-02-04T10:06:38.796139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def crop_image(img, tol=7):\n#     if img.ndim ==2:\n#         mask = img>tol\n#         return img[np.ix_(mask.any(1),mask.any(0))]\n#     elif img.ndim==3:\n#         gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n#         mask = gray_img>tol\n#         check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n#         if (check_shape == 0): # image is too dark so that we crop out everything,\n#             return img # return original image\n#         else:\n#             img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n#             img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n#             img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n#             img = np.stack([img1,img2,img3],axis=-1)\n            \n#         return img\n\n# def circle_crop(img):\n#     img = crop_image(img)\n\n#     height, width, depth = img.shape\n#     largest_side = np.max((height, width))\n#     img = cv2.resize(img, (largest_side, largest_side))\n\n#     height, width, depth = img.shape\n\n#     x = width//2\n#     y = height//2\n#     r = np.amin((x, y))\n\n#     circle_img = np.zeros((height, width), np.uint8)\n#     cv2.circle(circle_img, (x, y), int(r), 1, thickness=-1)\n#     img = cv2.bitwise_and(img, img, mask=circle_img)\n#     img = crop_image(img)\n\n#     return img\n    \n# def apply_preprocessing(img_path, sigmaX=10, clip_limit=5.0, tile_grid_size=(8, 8), target_size=(224, 224)):\n#     image = cv2.imread(img_path)\n#     image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n#     image = circle_crop(image)\n# #     image = cv2.resize(image, target_size)\n#     image = cv2.addWeighted(image,4, cv2.GaussianBlur(img , (0,0) , 30) ,-4 ,128)\n# #     image = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)\n#     image = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n#     image = cv2.resize(img, (IMG_SIZE,IMG_SIZE))\n\n# img_t = cv2.addWeighted(img,4, cv2.GaussianBlur(img , (0,0) , 30) ,-4 ,128)\n# #     clahe = cv2.createCLAHE(clipLimit=clip_limit, tileGridSize=tile_grid_size)\n# #     image = clahe.apply(image)\n# #     mage = cv2.addWeighted(image, 4, cv2.GaussianBlur(image, (0,0), sigmaX), -4 , 128)\n\n#     return image","metadata":{"execution":{"iopub.status.busy":"2024-02-04T10:52:49.70644Z","iopub.execute_input":"2024-02-04T10:52:49.7068Z","iopub.status.idle":"2024-02-04T10:52:49.721661Z","shell.execute_reply.started":"2024-02-04T10:52:49.706769Z","shell.execute_reply":"2024-02-04T10:52:49.720681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''This Function performs image processing on top of images by performing Gaussian Blur and Circle Crop'''\n\ndef crop_image_from_gray(img,tol=7):\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        \n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n    #         print(img1.shape,img2.shape,img3.shape)\n            img = np.stack([img1,img2,img3],axis=-1)\n#             print(img.shape)\n        return img\n    \n    \ndef circle_crop(img, sigmaX):   \n    \"\"\"Create circular crop around image centre\"\"\"    \n    img = crop_image_from_gray(img)    \n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    \n    height, width, depth = img.shape    \n    \n    x = int(width/2)\n    y = int(height/2)\n    r = np.amin((x,y))\n    \n    circle_img = np.zeros((height, width), np.uint8)\n    cv2.circle(circle_img, (x,y), int(r), 1, thickness=-1)\n    img = cv2.bitwise_and(img, img, mask=circle_img)\n    img = crop_image_from_gray(img)\n    img=cv2.addWeighted(img,4, cv2.GaussianBlur( img , (0,0) , sigmaX) ,-4 ,128)\n    return img \n\ndef apply_preprocessing(img_path, sigmaX=10, clip_limit=5.0, tile_grid_size=(8, 8), target_size=(224, 224)):\n    img = cv2.imread(img_path)\n    image = circle_crop(img,sigmaX = 30)\n#     image = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)\n\n    return image\n","metadata":{"execution":{"iopub.status.busy":"2024-02-04T16:24:23.493736Z","iopub.execute_input":"2024-02-04T16:24:23.494671Z","iopub.status.idle":"2024-02-04T16:24:23.511286Z","shell.execute_reply.started":"2024-02-04T16:24:23.494621Z","shell.execute_reply":"2024-02-04T16:24:23.510264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\n\ndef visualize_preprocessing(img_path, sigmaX=30, clip_limit=3.0, tile_grid_size=(8, 8), target_size=(224, 224)):\n    # Read the original image\n    img_original = cv2.imread(img_path)\n    \n    # Apply preprocessing to the image\n    img_preprocessed = apply_preprocessing(img_path, sigmaX, clip_limit, tile_grid_size, target_size)\n\n    # Visualize the original and preprocessed images\n    plt.figure(figsize=(12, 6))\n\n    # Original Image\n    plt.subplot(1, 2, 1)\n    plt.imshow(cv2.cvtColor(img_original, cv2.COLOR_BGR2RGB))\n    plt.title('Original Image')\n\n    # Preprocessed Image\n    plt.subplot(1, 2, 2)\n    plt.imshow(img_preprocessed, cmap='gray')  # Assuming preprocessed image is grayscale\n    plt.title('Preprocessed Image')\n\n    plt.show()\n\n# Example usage\nimg_path = '/kaggle/input/aptos2019-blindness-detection/train_images/00f6c1be5a33.png'\nvisualize_preprocessing(img_path)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-04T16:24:41.336896Z","iopub.execute_input":"2024-02-04T16:24:41.337839Z","iopub.status.idle":"2024-02-04T16:24:42.640796Z","shell.execute_reply.started":"2024-02-04T16:24:41.337804Z","shell.execute_reply":"2024-02-04T16:24:42.639754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-02-04T10:57:53.060616Z","iopub.execute_input":"2024-02-04T10:57:53.060969Z","iopub.status.idle":"2024-02-04T10:57:53.07533Z","shell.execute_reply.started":"2024-02-04T10:57:53.060944Z","shell.execute_reply":"2024-02-04T10:57:53.074331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\n# ...\n\n# Function to apply preprocessing and save images to corresponding folders\ndef apply_preprocessing_and_save(row):\n    img_path = row['file_path']  # Adjust column name based on your CSV\n    diagnosis_grade = row['diagnosis']\n    output_folder = os.path.join(base_directory, f'grade_{diagnosis_grade}')\n\n    # Apply preprocessing (replace this with your actual preprocessing logic)\n    img_array = apply_preprocessing(img_path)\n\n    # Convert NumPy array to Pillow Image\n    img = Image.fromarray((img_array * 255).astype(np.uint8))\n\n    # Save the preprocessed image\n    img.save(os.path.join(output_folder, os.path.basename(img_path)))","metadata":{"execution":{"iopub.status.busy":"2024-02-04T16:24:44.342455Z","iopub.execute_input":"2024-02-04T16:24:44.343401Z","iopub.status.idle":"2024-02-04T16:24:44.349904Z","shell.execute_reply.started":"2024-02-04T16:24:44.343364Z","shell.execute_reply":"2024-02-04T16:24:44.348763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.apply(apply_preprocessing_and_save, axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-02-04T16:24:49.996381Z","iopub.execute_input":"2024-02-04T16:24:49.996745Z","iopub.status.idle":"2024-02-04T18:36:16.492114Z","shell.execute_reply.started":"2024-02-04T16:24:49.996716Z","shell.execute_reply":"2024-02-04T18:36:16.491145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import zipfile\nimport os\n\n# Directory of the files to be zipped\ndirectory_to_zip = '/kaggle/working/'\n\n# Output ZIP file\nzip_file_path = '/kaggle/working/preprocessed_dataset.zip'\n\nwith zipfile.ZipFile(zip_file_path, 'w') as zip_file:\n    for foldername, subfolders, filenames in os.walk(directory_to_zip):\n        for filename in filenames:\n            file_path = os.path.join(foldername, filename)\n            arcname = os.path.relpath(file_path, directory_to_zip)\n            zip_file.write(file_path, arcname)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-04T18:41:37.198064Z","iopub.execute_input":"2024-02-04T18:41:37.19843Z","iopub.status.idle":"2024-02-04T18:42:04.468067Z","shell.execute_reply.started":"2024-02-04T18:41:37.198402Z","shell.execute_reply":"2024-02-04T18:42:04.466864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data, validation_data = train_test_split(train_data, test_size=0.2, stratify=train_data['diagnosis'], random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-02-04T18:36:16.494129Z","iopub.execute_input":"2024-02-04T18:36:16.494438Z","iopub.status.idle":"2024-02-04T18:36:17.3205Z","shell.execute_reply.started":"2024-02-04T18:36:16.494412Z","shell.execute_reply":"2024-02-04T18:36:17.319104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\nfrom keras.applications.vgg16 import VGG16, preprocess_input, decode_predictions\nfrom keras.layers import GlobalAveragePooling2D\nfrom keras.models import Model\n\n# Define the base directory where your grade folders are located\nbase_directory = '/kaggle/working/preprocessed_images/'\n\n# Define the image size expected by VGG16\nimg_size = (224, 224)\n\n# Create an ImageDataGenerator for training\ntrain_datagen = ImageDataGenerator(\n    preprocessing_function=preprocess_input,\n    rescale=1./255,  # Adding rescaling\n    validation_split=0.2,  # Adjust the validation split as needed\n    fill_mode='nearest'  # Fill mode for newly created pixels\n)","metadata":{"execution":{"iopub.status.busy":"2024-02-04T18:43:20.287458Z","iopub.execute_input":"2024-02-04T18:43:20.287878Z","iopub.status.idle":"2024-02-04T18:43:20.294729Z","shell.execute_reply.started":"2024-02-04T18:43:20.287846Z","shell.execute_reply":"2024-02-04T18:43:20.293628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator = train_datagen.flow_from_directory(\n    directory=base_directory,\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode='sparse',  # Assuming you have one-hot encoded labels\n    subset='training'  # Use the training subset\n)\n\n# Create the validation data generator\nvalidation_generator = train_datagen.flow_from_directory(\n    directory=base_directory,\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode='sparse',  # Assuming you have one-hot encoded labels\n    subset='validation'  # Use the validation subset\n)","metadata":{"execution":{"iopub.status.busy":"2024-02-04T18:43:35.951155Z","iopub.execute_input":"2024-02-04T18:43:35.951894Z","iopub.status.idle":"2024-02-04T18:43:36.113755Z","shell.execute_reply.started":"2024-02-04T18:43:35.951858Z","shell.execute_reply":"2024-02-04T18:43:36.112912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming 'train_generator' is your data generator\n# Print a few examples of labels to check their format\n\nfor _ in range(5):  # Print labels for the first 5 batches\n    batch_features, batch_labels = train_generator.next()\n    print(\"Batch Labels:\")\n    print(batch_labels)\n    print(\"-\" * 30)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-04T18:43:40.148518Z","iopub.execute_input":"2024-02-04T18:43:40.148905Z","iopub.status.idle":"2024-02-04T18:43:53.218023Z","shell.execute_reply.started":"2024-02-04T18:43:40.148875Z","shell.execute_reply":"2024-02-04T18:43:53.216938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.layers import Dense\nfrom tensorflow.keras.optimizers import Adam\n\n# Load VGG16 model without the top classification layer\nbase_model = VGG16(weights='imagenet', include_top=False, input_shape=(224, 224, 3))\n\n# Add a global average pooling layer\nx = base_model.output\nx = GlobalAveragePooling2D()(x)\n\n# Add a Dense layer for classification with softmax activation\nnum_classes = 5  # Update based on the number of classes in your classification task\npredictions = Dense(num_classes, activation='softmax')(x)\n\n# Create the final model for prediction\nmodel = Model(inputs=base_model.input, outputs=predictions)\n\n# Compile the model\n# optimizer = Adam(learning_rate=0.001) \noptimizer = Adam(learning_rate=0.001)\nmodel.compile(optimizer=optimizer, loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n","metadata":{"execution":{"iopub.status.busy":"2024-02-04T18:43:55.923503Z","iopub.execute_input":"2024-02-04T18:43:55.924253Z","iopub.status.idle":"2024-02-04T18:43:59.557562Z","shell.execute_reply.started":"2024-02-04T18:43:55.92422Z","shell.execute_reply":"2024-02-04T18:43:59.556545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compile the model (you may need to add your own final classification layer)\n\n# Train the model\nmodel.fit(\n    train_generator,\n    steps_per_epoch=len(train_generator),\n    epochs=10,  # Adjust the number of epochs\n    validation_data=validation_generator,\n    validation_steps=len(validation_generator)\n)","metadata":{"execution":{"iopub.status.busy":"2024-02-04T18:44:02.521871Z","iopub.execute_input":"2024-02-04T18:44:02.522489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **2.1 Class Distribution**","metadata":{}},{"cell_type":"code","source":"'''This Function Plots a Bar plot of output Classes Distribution'''\n\ndef plot_classes(df):\n    df_group = pd.DataFrame(df.groupby('diagnosis').agg('size').reset_index())\n    df_group.columns = ['diagnosis','count']\n\n    sns.set(rc={'figure.figsize':(10,5)}, style = 'whitegrid')\n    sns.barplot(x = 'diagnosis',y='count',data = df_group,palette = \"Blues_d\")\n    plt.title('Output Class Distribution')\n    plt.show() ","metadata":{"execution":{"iopub.status.busy":"2024-02-04T07:01:07.158851Z","iopub.execute_input":"2024-02-04T07:01:07.159707Z","iopub.status.idle":"2024-02-04T07:01:07.165586Z","shell.execute_reply.started":"2024-02-04T07:01:07.159676Z","shell.execute_reply":"2024-02-04T07:01:07.164609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_classes(df_train)","metadata":{"execution":{"iopub.status.busy":"2024-02-04T07:01:11.383205Z","iopub.execute_input":"2024-02-04T07:01:11.383553Z","iopub.status.idle":"2024-02-04T07:01:11.639161Z","shell.execute_reply.started":"2024-02-04T07:01:11.383527Z","shell.execute_reply":"2024-02-04T07:01:11.638283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Plot Summary - As we can see, there is class imbalance in the output class distribution. We shall account for this while training the models using data augmentation / class balancing methods.","metadata":{}},{"cell_type":"markdown","source":"### **2.2 Visualizing Images**","metadata":{}},{"cell_type":"code","source":"# Defining a global variable to be used as Image size..\nIMG_SIZE = 224","metadata":{"execution":{"iopub.status.busy":"2024-02-04T07:01:15.354269Z","iopub.execute_input":"2024-02-04T07:01:15.354609Z","iopub.status.idle":"2024-02-04T07:01:15.35882Z","shell.execute_reply.started":"2024-02-04T07:01:15.354583Z","shell.execute_reply":"2024-02-04T07:01:15.357897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''This Function converts a color image to gray scale image'''\n\ndef conv_gray(img):\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n    img = cv2.resize(img, (IMG_SIZE,IMG_SIZE))\n    return img\n\ndef extract_green_channel(img):\n    green_channel = img[:, :, 1]  # Extract the green channel (0: blue, 1: green, 2: red)\n    green_channel = cv2.resize(green_channel, (IMG_SIZE, IMG_SIZE))\n    return green_channel\n  \n    \n'''This Function shows the visual Image photo of 'n x 5' points (5 of each class)'''\n\ndef visualize_imgs(df,pts_per_class,color_scale):\n    df = df.groupby('diagnosis',group_keys = False).apply(lambda df: df.sample(pts_per_class))\n    df = df.reset_index(drop = True)\n    \n    plt.rcParams[\"axes.grid\"] = False\n    for pt in range(pts_per_class):\n        f, axarr = plt.subplots(1,5,figsize = (15,15))\n        axarr[0].set_ylabel(\"Sample Data Points\")\n        \n        df_temp = df[df.index.isin([pt + (pts_per_class*0),pt + (pts_per_class*1), pt + (pts_per_class*2),pt + (pts_per_class*3),pt + (pts_per_class*4)])]\n        for i in range(5):\n            if color_scale == 'gray':\n                img = extract_green_channel(cv2.imread(df_temp.file_path.iloc[i]))\n                axarr[i].imshow(img,cmap = color_scale)\n            else:\n                axarr[i].imshow(Image.open(df_temp.file_path.iloc[i]).resize((IMG_SIZE,IMG_SIZE)))\n            axarr[i].set_xlabel('Class '+str(df_temp.diagnosis.iloc[i]))\n\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-04T07:01:20.841618Z","iopub.execute_input":"2024-02-04T07:01:20.842572Z","iopub.status.idle":"2024-02-04T07:01:20.853401Z","shell.execute_reply.started":"2024-02-04T07:01:20.842538Z","shell.execute_reply":"2024-02-04T07:01:20.852222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Time Pass\n","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport cv2\n\ndef extract_green_channel(img):\n    green_channel = img[:, :, 1]  # Extract the green channel (0: blue, 1: green, 2: red)\n    return green_channel\n\ndef visualize_image_comparison(file_path):\n    # Load the color image\n    color_img = cv2.imread(file_path)\n\n    # Extract the green channel\n    green_channel_img = extract_green_channel(color_img)\n\n    # Plotting the original and green channel images side by side\n    plt.figure(figsize=(10, 5))\n\n    # Original Image\n    plt.subplot(1, 2, 1)\n    plt.imshow(cv2.cvtColor(color_img, cv2.COLOR_BGR2RGB))\n    plt.title('Original Image')\n    plt.axis('off')\n\n    # Green Channel Image with 'viridis' colormap\n    plt.subplot(1, 2, 2)\n    plt.imshow(green_channel_img, cmap='viridis')  # Use 'viridis' colormap for a greenish tint\n    plt.title('Green Channel')\n    plt.axis('off')\n\n    plt.show()\n\n\n# Example usage\nfile_path_to_visualize = '/kaggle/input/aptos2019-blindness-detection/train_images/000c1434d8d7.png'\nvisualize_image_comparison(file_path_to_visualize)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-04T07:01:30.706324Z","iopub.execute_input":"2024-02-04T07:01:30.707143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_imgs(df_train,3,color_scale = None)","metadata":{"execution":{"iopub.status.busy":"2024-02-04T07:01:42.051252Z","iopub.execute_input":"2024-02-04T07:01:42.051939Z","iopub.status.idle":"2024-02-04T07:01:48.141305Z","shell.execute_reply.started":"2024-02-04T07:01:42.051907Z","shell.execute_reply":"2024-02-04T07:01:48.140405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_imgs(df_train,2,color_scale = 'gray')","metadata":{"execution":{"iopub.status.busy":"2024-02-04T07:01:48.143024Z","iopub.execute_input":"2024-02-04T07:01:48.143325Z","iopub.status.idle":"2024-02-04T07:01:52.946175Z","shell.execute_reply.started":"2024-02-04T07:01:48.1433Z","shell.execute_reply":"2024-02-04T07:01:52.945154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Plot Summary - As we can see, as we move towards higher classes, we are able to comprehend larger number of abnormalities in the eye images. Also, the lightning and brightness conditions are not even across all images. We will try to handle this using image processing techniques. Also, Gray Scale Images are giving better visualization of the eye features as compared to RGB images.\n\nThis below photo shows what the eye diabetic retinopathy condition refers to:\n![](http://sa1s3optim.patientpop.com/assets/images/provider/photos/1947516.jpeg)","metadata":{}},{"cell_type":"markdown","source":"## **3. Image Processing**","metadata":{}},{"cell_type":"markdown","source":"### **3.1 Gaussian Blur**","metadata":{}},{"cell_type":"code","source":"'''This section of code applies gaussian blur on top of image'''\n\nrn = np.random.randint(low = 0,high = len(df_train) - 1)\n\nimg = cv2.imread(df_train.file_path.iloc[rn])\nimg = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\nimg = cv2.resize(img, (IMG_SIZE,IMG_SIZE))\n\nimg_t = cv2.addWeighted(img,4, cv2.GaussianBlur(img , (0,0) , 30) ,-4 ,128)\n\nf, axarr = plt.subplots(1,2,figsize = (11,11))\naxarr[0].imshow(img)\naxarr[1].imshow(img_t)\nplt.title('After applying Gaussian Blur')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-04T07:01:59.147018Z","iopub.execute_input":"2024-02-04T07:01:59.14785Z","iopub.status.idle":"2024-02-04T07:01:59.914063Z","shell.execute_reply.started":"2024-02-04T07:01:59.147821Z","shell.execute_reply":"2024-02-04T07:01:59.91321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Plot Summary - As we can see, after applying Gaussian Blur, We are able to bring out the features/image details much more clearer in the eye.","metadata":{}},{"cell_type":"markdown","source":"### **3.2 Gaussian Blur with Circular Cropping**","metadata":{}},{"cell_type":"code","source":"'''This Function performs image processing on top of images by performing Gaussian Blur and Circle Crop'''\n\ndef crop_image_from_gray(img,tol=7):\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        \n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n    #         print(img1.shape,img2.shape,img3.shape)\n            img = np.stack([img1,img2,img3],axis=-1)\n            print(img.shape)\n        return img\n    \n    \ndef circle_crop(img, sigmaX):   \n    \"\"\"Create circular crop around image centre\"\"\"    \n    img = crop_image_from_gray(img)    \n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    \n    height, width, depth = img.shape    \n    \n    x = int(width/2)\n    y = int(height/2)\n    r = np.amin((x,y))\n    \n    circle_img = np.zeros((height, width), np.uint8)\n    cv2.circle(circle_img, (x,y), int(r), 1, thickness=-1)\n    img = cv2.bitwise_and(img, img, mask=circle_img)\n    img = crop_image_from_gray(img)\n    img=cv2.addWeighted(img,4, cv2.GaussianBlur( img , (0,0) , sigmaX) ,-4 ,128)\n    return img ","metadata":{"execution":{"iopub.status.busy":"2024-02-04T10:50:45.563435Z","iopub.execute_input":"2024-02-04T10:50:45.563919Z","iopub.status.idle":"2024-02-04T10:50:45.577793Z","shell.execute_reply.started":"2024-02-04T10:50:45.563878Z","shell.execute_reply":"2024-02-04T10:50:45.576694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''Perform Image Processing on a sample image'''\n\nrn = np.random.randint(low = 0,high = len(df_train) - 1)\n\n#img = img_t\nimg = cv2.imread(df_train.file_path.iloc[rn])\nimg_t = circle_crop(img,sigmaX = 30)\n\nf, axarr = plt.subplots(1,2,figsize = (11,11))\naxarr[0].imshow(cv2.resize(cv2.cvtColor(img, cv2.COLOR_BGR2RGB),(IMG_SIZE,IMG_SIZE)))\naxarr[1].imshow(img_t)\nplt.title('After applying Circular Crop and Gaussian Blur')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-04T10:50:50.051658Z","iopub.execute_input":"2024-02-04T10:50:50.052521Z","iopub.status.idle":"2024-02-04T10:50:51.134484Z","shell.execute_reply.started":"2024-02-04T10:50:50.052486Z","shell.execute_reply":"2024-02-04T10:50:51.13356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''This Function shows the visual Image photo of 'n x 5' points (5 of each class) \nand performs image processing (Gaussian Blur, Circular crop) transformation on top of that'''\n\ndef visualize_img_process(df,pts_per_class,sigmaX):\n    df = df.groupby('diagnosis',group_keys = False).apply(lambda df: df.sample(pts_per_class))\n    df = df.reset_index(drop = True)\n    \n    plt.rcParams[\"axes.grid\"] = False\n    for pt in range(pts_per_class):\n        f, axarr = plt.subplots(1,5,figsize = (15,15))\n        axarr[0].set_ylabel(\"Sample Data Points\")\n        \n        df_temp = df[df.index.isin([pt + (pts_per_class*0),pt + (pts_per_class*1), pt + (pts_per_class*2),pt + (pts_per_class*3),pt + (pts_per_class*4)])]\n        for i in range(5):\n            img = cv2.imread(df_temp.file_path.iloc[i])\n            img = circle_crop(img,sigmaX)\n#             img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n            axarr[i].imshow(img)\n            axarr[i].set_xlabel('Class '+str(df_temp.diagnosis.iloc[i]))\n\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-04T10:46:11.996028Z","iopub.execute_input":"2024-02-04T10:46:11.9964Z","iopub.status.idle":"2024-02-04T10:46:12.005542Z","shell.execute_reply.started":"2024-02-04T10:46:11.99637Z","shell.execute_reply":"2024-02-04T10:46:12.0044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_img_process(df_train,5,sigmaX = 30)","metadata":{"execution":{"iopub.status.busy":"2024-02-04T10:46:15.727935Z","iopub.execute_input":"2024-02-04T10:46:15.728744Z","iopub.status.idle":"2024-02-04T10:46:16.617064Z","shell.execute_reply.started":"2024-02-04T10:46:15.72871Z","shell.execute_reply":"2024-02-04T10:46:16.615737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\n\ndef apply_clahe(img, clip_limit=2.0, tile_grid_size=(8, 8)):\n    \"\"\"\n    Apply CLAHE to the given image.\n\n    Parameters:\n    - img: Input image (grayscale).\n    - clip_limit: Threshold for contrast limiting.\n    - tile_grid_size: Size of the grid for histogram equalization.\n\n    Returns:\n    - img_clahe: Image after applying CLAHE.\n    \"\"\"\n    clahe = cv2.createCLAHE(clipLimit=clip_limit, tileGridSize=tile_grid_size)\n    img_clahe = clahe.apply(img)\n    return img_clahe\n\ndef visualize_img_clahe(df, pts_per_class, sigmaX, clip_limit=3.0, tile_grid_size=(8, 8)):\n    \"\"\"\n    Visualize images with CLAHE applied after Gaussian filtering.\n\n    Parameters:\n    - df: DataFrame containing image information.\n    - pts_per_class: Number of points to sample for each class.\n    - sigmaX: Parameter for Gaussian blur.\n    - clip_limit: Threshold for contrast limiting in CLAHE.\n    - tile_grid_size: Size of the grid for histogram equalization in CLAHE.\n    \"\"\"\n    df = df.groupby('diagnosis', group_keys=False).apply(lambda df: df.sample(pts_per_class))\n    df = df.reset_index(drop=True)\n\n    plt.rcParams[\"axes.grid\"] = False\n    for pt in range(pts_per_class):\n        f, axarr = plt.subplots(1, 5, figsize=(15, 15))\n        axarr[0].set_ylabel(\"Sample Data Points\")\n\n        df_temp = df[df.index.isin([pt + (pts_per_class * 0), pt + (pts_per_class * 1), pt + (pts_per_class * 2),\n                                     pt + (pts_per_class * 3), pt + (pts_per_class * 4)])]\n        for i in range(5):\n            img = cv2.imread(df_temp.file_path.iloc[i])\n            img = circle_crop(img, sigmaX)\n\n            # Convert to grayscale\n            img_gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n\n            # Apply Gaussian blur\n            img_blur = cv2.addWeighted(img_gray, 4, cv2.GaussianBlur(img_gray, (0, 0), sigmaX), -4, 128)\n\n            # Apply CLAHE to the Gaussian blurred image\n            img_clahe = apply_clahe(img_blur, clip_limit=clip_limit, tile_grid_size=tile_grid_size)\n\n            axarr[i].imshow(img_clahe, cmap='gray')\n            axarr[i].set_xlabel('Class ' + str(df_temp.diagnosis.iloc[i]))\n\n        plt.show()\n\n# Example usage with CLAHE parameters\nvisualize_img_clahe(df_train, pts_per_class=1, sigmaX=30, clip_limit=2.0, tile_grid_size=(8, 8))\n","metadata":{"execution":{"iopub.status.busy":"2024-02-04T10:43:30.648251Z","iopub.execute_input":"2024-02-04T10:43:30.648787Z","iopub.status.idle":"2024-02-04T10:43:36.134173Z","shell.execute_reply.started":"2024-02-04T10:43:30.648753Z","shell.execute_reply":"2024-02-04T10:43:36.133223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_img_clahe(df_train, pts_per_class=1, sigmaX=30, clip_limit=3.0, tile_grid_size=(8, 8))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_img_clahe(df_train, pts_per_class=1, sigmaX=30, clip_limit=4.0, tile_grid_size=(8, 8))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_img_clahe(df_train, pts_per_class=1, sigmaX=30, clip_limit=1.0, tile_grid_size=(4, 4))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG_SIZE = 224\n\ndef apply_preprocessing(img_path, sigmaX =30 , clip_limit=3.0, tile_grid_size=(8, 8)):\n    \"\"\"\n    Apply the same preprocessing steps as in visualize_img_clahe to an image.\n\n    Parameters:\n    - img_path: Path to the original image.\n    - sigmaX: Parameter for Gaussian blur.\n    - clip_limit: Threshold for contrast limiting in CLAHE.\n    - tile_grid_size: Size of the grid for histogram equalization in CLAHE.\n\n    Returns:\n    - img_preprocessed: Preprocessed image.\n    \"\"\"\n    img = cv2.imread(img_path)\n    img = circle_crop(img, sigmaX)\n\n    # Convert to grayscale\n    img_gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n\n    # Apply Gaussian blur\n    img_blur = cv2.addWeighted(img_gray, 4, cv2.GaussianBlur(img_gray, (0, 0), sigmaX), -4, 128)\n\n    # Apply CLAHE to the Gaussian blurred image\n    clahe = cv2.createCLAHE(clipLimit=clip_limit, tileGridSize=tile_grid_size)\n    img_clahe = clahe.apply(img_blur)\n\n    return img_clahe\n\n\n\n# Load data\n\ntrain, test = load_data()\n\n# Assuming you have a DataFrame 'train' with 'file_path' and 'diagnosis' columns\n\n# Create a new column for preprocessed images\ntrain['preprocessed_image'] = train['file_path'].apply(lambda x: apply_preprocessing(x))\n\n# Split the data into train and validation sets\ntrain_data, validation_data = train_test_split(train, test_size=0.2, stratify=train['diagnosis'], random_state=42)\n\n# Convert labels to one-hot encoding\nnum_classes = 5  # Update based on the number of classes\ntrain_data['diagnosis'] = to_categorical(train_data['diagnosis'], num_classes=num_classes)\nvalidation_data['diagnosis'] = to_categorical(validation_data['diagnosis'], num_classes=num_classes)\n\n# Custom Data Generator for preprocessed images\nclass CustomDataGenerator(ImageDataGenerator):\n    def __init__(self, *args, **kwargs):\n        super(CustomDataGenerator, self).__init__(*args, **kwargs)\n\n    def preprocess_image(self, img_path):\n        img = cv2.imread(img_path)\n        img = apply_preprocessing(img)  # Apply your preprocessing function here\n        img = img / 255.0  # Normalize pixel values to be between 0 and 1\n        return img\n\n    def flow_from_dataframe(self, dataframe, *args, **kwargs):\n        return super().flow_from_dataframe(\n            dataframe,\n            *args,\n            **kwargs,\n            x_col='preprocessed_image',\n            y_col='diagnosis',\n            target_size=(IMG_SIZE, IMG_SIZE),\n            batch_size=32,\n            class_mode='categorical'\n        )\n\n# Example Usage\ntrain_datagen = CustomDataGenerator(rescale=1./255)\n\n# Assuming 'preprocessed_image' column contains the paths to preprocessed images\ntrain_generator = train_datagen.flow_from_dataframe(\n    train_data,\n    x_col='file_path',\n    y_col='diagnosis',\n    target_size=(IMG_SIZE, IMG_SIZE),\n    batch_size=32,\n    class_mode='categorical'\n)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG_SIZE = 224\n\ndef apply_preprocessing(img_path, sigmaX =30 , clip_limit=3.0, tile_grid_size=(8, 8)):\n    \"\"\"\n    Apply the same preprocessing steps as in visualize_img_clahe to an image.\n\n    Parameters:\n    - img_path: Path to the original image.\n    - sigmaX: Parameter for Gaussian blur.\n    - clip_limit: Threshold for contrast limiting in CLAHE.\n    - tile_grid_size: Size of the grid for histogram equalization in CLAHE.\n\n    Returns:\n    - img_preprocessed: Preprocessed image.\n    \"\"\"\n    img = cv2.imread(img_path)\n    img = circle_crop(img, sigmaX)\n\n    # Convert to grayscale\n    img_gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n\n    # Apply Gaussian blur\n    img_blur = cv2.addWeighted(img_gray, 4, cv2.GaussianBlur(img_gray, (0, 0), sigmaX), -4, 128)\n\n    # Apply CLAHE to the Gaussian blurred image\n    clahe = cv2.createCLAHE(clipLimit=clip_limit, tileGridSize=tile_grid_size)\n    img_clahe = clahe.apply(img_blur)\n\n    return img_clahe","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load data\n\ntrain, test = load_data()\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming you have a DataFrame 'train' with 'file_path' and 'diagnosis' columns\n\n# Create a new column for preprocessed images\ntrain['preprocessed_image'] = train['file_path'].apply(lambda x: apply_preprocessing(x))\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(10)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sklearn\nimport keras\nimport keras.utils\nfrom keras import utils as np_utils\nfrom tensorflow.keras import utils as np_utils\nfrom sklearn.model_selection import train_test_split\nfrom keras.utils import to_categorical\n\n\n# Split the data into train and validation sets\ntrain_data, validation_data = train_test_split(train, test_size=0.2, stratify=train['diagnosis'], random_state=42)\n\n# Convert labels to one-hot encoding\nnum_classes = 5  # Update based on the number of classes\ntrain_data['diagnosis'] = to_categorical(train_data['diagnosis'], num_classes=num_classes)\nvalidation_data['diagnosis'] = to_categorical(validation_data['diagnosis'], num_classes=num_classes)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datagen = CustomDataGenerator(rescale=1./255)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(10)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"","metadata":{}},{"cell_type":"code","source":"train_generator = train_datagen.flow_from_dataframe(\n    train_data,\n    x_col='preprocessed_image',\n    y_col='diagnosis',\n    target_size=(IMG_SIZE, IMG_SIZE),\n    batch_size=32,\n    class_mode='categorical'\n)\n\nvalidation_generator = train_datagen.flow_from_dataframe(\n    validation_data,\n    x_col='preprocessed_image',\n    y_col='diagnosis',\n    target_size=(IMG_SIZE, IMG_SIZE),\n    batch_size=32,\n    class_mode='categorical'\n)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datagen = CustomDataGenerator(rescale=1./255)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Dropout, Flatten, Dense\nfrom tensorflow.keras.optimizers import Adam\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_cnn_model():\n    model = tf.keras.models.Sequential([\n        #CONV1\n        Conv2D(64, (3,3), activation='relu', padding ='same', input_shape=(224,224,3)),\n        Conv2D(64, (3,3), activation='relu', padding ='same'),\n        MaxPooling2D(2,2),\n        #Dropout(0.5),\n        #CONV2\n        Conv2D(128, (3,3), activation='relu', padding ='same'),\n        Conv2D(128, (3,3), activation='relu', padding ='same'),\n        MaxPooling2D(2,2),\n        #Dropout(0.5),\n        #CONV3\n        Conv2D(256, (3,3), activation='relu', padding ='same'),\n        Conv2D(256, (3,3), activation='relu', padding ='same'),\n        Conv2D(256, (3,3), activation='relu', padding ='same'),\n        Conv2D(256, (3,3), activation='relu', padding ='same'),\n        Conv2D(256, (3,3), activation='relu', padding ='same'),\n        MaxPooling2D(2,2),\n        #Dropout(0.5),\n        #CONV4\n        Conv2D(512, (3,3), activation='relu', padding ='same'),\n        Conv2D(512, (3,3), activation='relu', padding ='same'),\n        Conv2D(512, (3,3), activation='relu', padding ='same'),\n        Conv2D(512, (3,3), activation='relu', padding ='same'),\n        Conv2D(512, (3,3), activation='relu', padding ='same'),\n        MaxPooling2D(2,2),\n        #Dropout(0.5),\n        tf.keras.layers.Flatten(),\n        tf.keras.layers.Dense(1024, activation='relu'),\n        tf.keras.layers.Dense(5, activation='softmax')  # Assuming 5 classes\n    ])\n\n    model.compile(loss='categorical_crossentropy', optimizer=Adam(lr=0.0001), metrics=['accuracy'])\n    return model\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.model_selection import train_test_split\n\n# Assuming 'train' DataFrame contains file paths ('file_path') and labels ('diagnosis')\n# Processed images are stored in the paths specified in 'file_path'\n\n# Data Splitting\ntrain_df, validation_df = train_test_split(train, test_size=0.2, stratify=train['diagnosis'], random_state=42)\n\n# Data Augmentation\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\nvalidation_datagen = ImageDataGenerator(rescale=1./255)\n\n# Train Data Generator\ntrain_generator = train_datagen.flow_from_dataframe(\n    train_df,\n    x_col='file_path',\n    y_col='diagnosis',\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode='categorical'\n)\n\n# Validation Data Generator\nvalidation_generator = validation_datagen.flow_from_dataframe(\n    validation_df,\n    x_col='file_path',\n    y_col='diagnosis',\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode='categorical'\n)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cnn_model = create_cnn_model()\n\n# Model Training\nhistory = cnn_model.fit(train_generator, epochs=10, validation_data=validation_generator)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}