{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-15T07:06:10.233163Z","iopub.execute_input":"2023-04-15T07:06:10.233791Z","iopub.status.idle":"2023-04-15T07:06:14.391049Z","shell.execute_reply.started":"2023-04-15T07:06:10.233751Z","shell.execute_reply":"2023-04-15T07:06:14.389884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm,tqdm_notebook\nfrom prettytable import PrettyTable\nimport pickle\nimport os\nprint('CWD is ',os.getcwd())\n\n# Vis Libs..\nfrom sklearn.manifold import TSNE\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nplt.rcParams[\"axes.grid\"] = False\n\n# Image Libs.\nfrom PIL import Image\nimport cv2\n\n# DL Libs..\nimport keras\nfrom keras import applications\nfrom tensorflow.keras.utils import img_to_array\nfrom tensorflow.keras.utils import load_img\nfrom tensorflow.keras.utils import array_to_img\n# from tensorflow.keras.utils import ImageDataGenerator\n# from keras.preprocessing.image import ImageDataGenerator,img_to_array,array_to_img,load_img\nfrom keras import optimizers,Model,Sequential\nfrom keras.layers import Input,GlobalAveragePooling2D,Dropout,Dense,Activation\nfrom keras.callbacks import EarlyStopping,ReduceLROnPlateau","metadata":{"execution":{"iopub.status.busy":"2023-04-15T07:06:14.393247Z","iopub.execute_input":"2023-04-15T07:06:14.39391Z","iopub.status.idle":"2023-04-15T07:06:24.894737Z","shell.execute_reply.started":"2023-04-15T07:06:14.393857Z","shell.execute_reply":"2023-04-15T07:06:24.893429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nThis function reads data from the respective train and test directories\n'''\n\ndef load_data():\n    train = pd.read_csv('/kaggle/input/aptos2019-blindness-detection/train.csv')\n    test = pd.read_csv('/kaggle/input/aptos2019-blindness-detection/test.csv')\n    \n    train_dir = os.path.join('./','/kaggle/input/aptos2019-blindness-detection/train_images')\n    test_dir = os.path.join('./','/kaggle/input/aptos2019-blindness-detection/train_images')\n    \n    train['file_path'] = train['id_code'].map(lambda x: os.path.join(train_dir,'{}.png'.format(x)))\n    test['file_path'] = test['id_code'].map(lambda x: os.path.join(test_dir,'{}.png'.format(x)))\n    \n    train['file_name'] = train[\"id_code\"].apply(lambda x: x + \".png\")\n    test['file_name'] = test[\"id_code\"].apply(lambda x: x + \".png\")\n    \n    train['diagnosis'] = train['diagnosis'].astype(str)\n    \n    return train,test","metadata":{"execution":{"iopub.status.busy":"2023-04-15T07:06:24.900106Z","iopub.execute_input":"2023-04-15T07:06:24.903109Z","iopub.status.idle":"2023-04-15T07:06:24.915266Z","shell.execute_reply.started":"2023-04-15T07:06:24.903066Z","shell.execute_reply":"2023-04-15T07:06:24.913646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train,df_test = load_data()\nprint(df_train.shape,df_test.shape,'\\n')\ndf_train.head(6)","metadata":{"execution":{"iopub.status.busy":"2023-04-15T07:06:24.922478Z","iopub.execute_input":"2023-04-15T07:06:24.92385Z","iopub.status.idle":"2023-04-15T07:06:24.999996Z","shell.execute_reply.started":"2023-04-15T07:06:24.923806Z","shell.execute_reply":"2023-04-15T07:06:24.998159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''This Function Plots a Bar plot of output Classes Distribution'''\n\ndef plot_classes(df):\n    df_group = pd.DataFrame(df.groupby('diagnosis').agg('size').reset_index())\n    df_group.columns = ['diagnosis','count']\n\n    sns.set(rc={'figure.figsize':(10,5)}, style = 'whitegrid')\n    sns.barplot(x = 'diagnosis',y='count',data = df_group,palette = \"Blues_d\")\n    plt.title('Output Class Distribution')\n    plt.show() ","metadata":{"execution":{"iopub.status.busy":"2023-04-15T07:06:25.005909Z","iopub.execute_input":"2023-04-15T07:06:25.007926Z","iopub.status.idle":"2023-04-15T07:06:25.018282Z","shell.execute_reply.started":"2023-04-15T07:06:25.007878Z","shell.execute_reply":"2023-04-15T07:06:25.016994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_classes(df_train)","metadata":{"execution":{"iopub.status.busy":"2023-04-15T07:06:25.019655Z","iopub.execute_input":"2023-04-15T07:06:25.021112Z","iopub.status.idle":"2023-04-15T07:06:25.367121Z","shell.execute_reply.started":"2023-04-15T07:06:25.021063Z","shell.execute_reply":"2023-04-15T07:06:25.365937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defining a global variable to be used as Image size..\nIMG_SIZE = 200","metadata":{"execution":{"iopub.status.busy":"2023-04-15T07:06:25.36888Z","iopub.execute_input":"2023-04-15T07:06:25.36958Z","iopub.status.idle":"2023-04-15T07:06:25.374745Z","shell.execute_reply.started":"2023-04-15T07:06:25.369538Z","shell.execute_reply":"2023-04-15T07:06:25.373271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''This Function converts a color image to gray scale image'''\n\ndef conv_gray(img):\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n    img = cv2.resize(img, (IMG_SIZE,IMG_SIZE))\n    return img\n  \n    \n'''\nThis Function shows the visual Image photo of 'n x 5' points (5 of each class)\n'''\n\ndef visualize_imgs(df,pts_per_class,color_scale):\n    df = df.groupby('diagnosis',group_keys = False).apply(lambda df: df.sample(pts_per_class))\n    df = df.reset_index(drop = True)\n    \n    plt.rcParams[\"axes.grid\"] = False\n    for pt in range(pts_per_class):\n        f, axarr = plt.subplots(1,5,figsize = (15,15))\n        axarr[0].set_ylabel(\"Sample Data Points\")\n        \n        df_temp = df[df.index.isin([pt + (pts_per_class*0),pt + (pts_per_class*1), pt + (pts_per_class*2),pt + (pts_per_class*3),pt + (pts_per_class*4)])]\n        for i in range(5):\n            if color_scale == 'gray':\n                img = conv_gray(cv2.imread(df_temp.file_path.iloc[i]))\n                axarr[i].imshow(img,cmap = color_scale)\n            else:\n                axarr[i].imshow(Image.open(df_temp.file_path.iloc[i]).resize((IMG_SIZE,IMG_SIZE)))\n            axarr[i].set_xlabel('Class '+str(df_temp.diagnosis.iloc[i]))\n\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-15T07:06:25.376235Z","iopub.execute_input":"2023-04-15T07:06:25.377649Z","iopub.status.idle":"2023-04-15T07:06:25.389745Z","shell.execute_reply.started":"2023-04-15T07:06:25.377588Z","shell.execute_reply":"2023-04-15T07:06:25.38848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_imgs(df_train,3,color_scale = None)","metadata":{"execution":{"iopub.status.busy":"2023-04-15T07:06:25.391459Z","iopub.execute_input":"2023-04-15T07:06:25.392003Z","iopub.status.idle":"2023-04-15T07:06:30.76028Z","shell.execute_reply.started":"2023-04-15T07:06:25.391935Z","shell.execute_reply":"2023-04-15T07:06:30.759299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_imgs(df_train,2,color_scale = 'gray')","metadata":{"execution":{"iopub.status.busy":"2023-04-15T07:06:30.76414Z","iopub.execute_input":"2023-04-15T07:06:30.764774Z","iopub.status.idle":"2023-04-15T07:06:33.55518Z","shell.execute_reply.started":"2023-04-15T07:06:30.764736Z","shell.execute_reply":"2023-04-15T07:06:33.554107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nThis section of code applies gaussian blur on top of image\n'''\n\nrn = np.random.randint(low = 0,high = len(df_train) - 1)\n\nimg = cv2.imread(df_train.file_path.iloc[rn])\nimg = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\nimg = cv2.resize(img, (IMG_SIZE,IMG_SIZE))\n\nimg_t = cv2.addWeighted(img,4, cv2.GaussianBlur(img , (0,0) , 30) ,-4 ,128)\n\nf, axarr = plt.subplots(1,2,figsize = (11,11))\naxarr[0].imshow(img)\naxarr[1].imshow(img_t)\nplt.title('After applying Gaussian Blur')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-15T07:06:33.556866Z","iopub.execute_input":"2023-04-15T07:06:33.557508Z","iopub.status.idle":"2023-04-15T07:06:34.302941Z","shell.execute_reply.started":"2023-04-15T07:06:33.557467Z","shell.execute_reply":"2023-04-15T07:06:34.299096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nThis Function performs image processing on top of images by performing Gaussian Blur and Circle Crop\n'''\n\ndef crop_image_from_gray(img,tol=7):\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        \n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n    #         print(img1.shape,img2.shape,img3.shape)\n            img = np.stack([img1,img2,img3],axis=-1)\n    #         print(img.shape)\n        return img\n    \n    \ndef circle_crop(img, sigmaX):   \n    \"\"\"\n    Create circular crop around image centre    \n    \"\"\"    \n    img = crop_image_from_gray(img)    \n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    \n    height, width, depth = img.shape    \n    \n    x = int(width/2)\n    y = int(height/2)\n    r = np.amin((x,y))\n    \n    circle_img = np.zeros((height, width), np.uint8)\n    cv2.circle(circle_img, (x,y), int(r), 1, thickness=-1)\n    img = cv2.bitwise_and(img, img, mask=circle_img)\n    img = crop_image_from_gray(img)\n    img=cv2.addWeighted(img,4, cv2.GaussianBlur( img , (0,0) , sigmaX) ,-4 ,128)\n    return img ","metadata":{"execution":{"iopub.status.busy":"2023-04-15T07:06:34.304391Z","iopub.execute_input":"2023-04-15T07:06:34.305385Z","iopub.status.idle":"2023-04-15T07:06:34.32027Z","shell.execute_reply.started":"2023-04-15T07:06:34.305344Z","shell.execute_reply":"2023-04-15T07:06:34.319242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''Perform Image Processing on a sample image'''\n\nrn = np.random.randint(low = 0,high = len(df_train) - 1)\n\n#img = img_t\nimg = cv2.imread(df_train.file_path.iloc[rn])\nimg_t = circle_crop(img,sigmaX = 30)\n\nf, axarr = plt.subplots(1,2,figsize = (11,11))\naxarr[0].imshow(cv2.resize(cv2.cvtColor(img, cv2.COLOR_BGR2RGB),(IMG_SIZE,IMG_SIZE)))\naxarr[1].imshow(img_t)\nplt.title('After applying Circular Crop and Gaussian Blur')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-15T07:06:34.322033Z","iopub.execute_input":"2023-04-15T07:06:34.322548Z","iopub.status.idle":"2023-04-15T07:06:35.453218Z","shell.execute_reply.started":"2023-04-15T07:06:34.322508Z","shell.execute_reply":"2023-04-15T07:06:35.448828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nThis Function shows the visual Image photo of 'n x 5' points (5 of each class) \nand performs image processing (Gaussian Blur, Circular crop) transformation on top of that\n'''\n\ndef visualize_img_process(df,pts_per_class,sigmaX):\n    df = df.groupby('diagnosis',group_keys = False).apply(lambda df: df.sample(pts_per_class))\n    df = df.reset_index(drop = True)\n    \n    plt.rcParams[\"axes.grid\"] = False\n    for pt in range(pts_per_class):\n        f, axarr = plt.subplots(1,5,figsize = (15,15))\n        axarr[0].set_ylabel(\"Sample Data Points\")\n        \n        df_temp = df[df.index.isin([pt + (pts_per_class*0),pt + (pts_per_class*1), pt + (pts_per_class*2),pt + (pts_per_class*3),pt + (pts_per_class*4)])]\n        for i in range(5):\n            img = cv2.imread(df_temp.file_path.iloc[i])\n            img = circle_crop(img,sigmaX)\n            axarr[i].imshow(img)\n            axarr[i].set_xlabel('Class '+str(df_temp.diagnosis.iloc[i]))\n\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-15T07:06:35.454727Z","iopub.execute_input":"2023-04-15T07:06:35.455588Z","iopub.status.idle":"2023-04-15T07:06:35.466226Z","shell.execute_reply.started":"2023-04-15T07:06:35.455549Z","shell.execute_reply":"2023-04-15T07:06:35.464923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_img_process(df_train,5,sigmaX = 30)","metadata":{"execution":{"iopub.status.busy":"2023-04-15T07:06:35.469513Z","iopub.execute_input":"2023-04-15T07:06:35.470529Z","iopub.status.idle":"2023-04-15T07:07:16.443068Z","shell.execute_reply.started":"2023-04-15T07:06:35.470488Z","shell.execute_reply":"2023-04-15T07:07:16.442127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train image data\nnpix = 224 # resize to npix x npix (for now)\nX_train = np.zeros((df_train.shape[0], npix, npix))\nfor i in tqdm_notebook(range(df_train.shape[0])):\n    # load an image\n    img = cv2.imread(df_train.file_path.iloc[i])\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY) \n    X_train[i, :, :] = cv2.resize(img, (npix, npix)) \n    \nprint(\"X_train shape: \" + str(np.shape(X_train)))  ","metadata":{"execution":{"iopub.status.busy":"2023-04-15T07:07:16.444345Z","iopub.execute_input":"2023-04-15T07:07:16.445892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# normalize\nX = X_train / 255\n\n# reshape\nX = X.reshape(X.shape[0], -1)\ntrainy = df_train['diagnosis']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"per_vals = [2,5,10,15,20,30,40,50]\n\nfor per in tqdm_notebook(per_vals):\n    X_decomposed = TSNE(n_components=2,perplexity = per).fit_transform(X)\n    df_tsne = pd.DataFrame(data=X_decomposed, columns=['Dimension_x','Dimension_y'])\n    df_tsne['Score'] = trainy.values\n    \n    sns.FacetGrid(df_tsne, hue='Score', size=6).map(plt.scatter, 'Dimension_x', 'Dimension_y').add_legend()\n    plt.title('TSNE for perplexity = ' + str(per))\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ref - https://www.youtube.com/watch?v=hxLU32zhze0\n# ref - https://stackoverflow.com/questions/49643907/clipping-input-data-to-the-valid-range-for-imshow-with-rgb-data-0-1-for-floa\n# ref - https://keras.io/preprocessing/image/\n\n'''This Function generates 'lim' number of Image Augmentations from a random Image in the directory'''\n\ndef generate_augmentations(lim):\n    datagen = ImageDataGenerator(featurewise_center=True,\n                                 featurewise_std_normalization=True,\n                                 rotation_range=20,\n                                 #width_shift_range=0.2,\n                                 #height_shift_range=0.2,\n                                 horizontal_flip=True)\n    img = cv2.imread(df_train.file_path.iloc[np.random.randint(low = 0,high = len(df_train) - 1)])\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = cv2.resize(img, (IMG_SIZE,IMG_SIZE))\n    plt.imshow(img)\n    plt.title('ORIGINAL IMAGE')\n    plt.show()\n    \n    img_arr = img.reshape((1,) + img.shape)\n    \n    i = 0\n    for img_iterator in datagen.flow(x = img_arr,batch_size = 1):\n        i = i + 1\n        if i > lim:\n            break\n        plt.imshow((img_iterator.reshape(img_arr[0].shape)).astype(np.uint8))\n        plt.title('IMAGE AUGMENTATION ' + str(i))\n        plt.show() ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sklearn libs..\nfrom sklearn.model_selection import train_test_split\ndf_train_train,df_train_valid = train_test_split(df_train,test_size = 0.2)\nprint(df_train_train.shape,df_train_valid.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef plot_classes(df,title):\n    df_group = pd.DataFrame(df.groupby('diagnosis').agg('size').reset_index())\n    df_group.columns = ['diagnosis','count']\n\n    sns.set(rc={'figure.figsize':(10,5)}, style = 'whitegrid')\n    sns.barplot(x = 'diagnosis',y='count',data = df_group,palette = \"Blues_d\")\n    plt.title('Output Class Distribution ' + str(title))\n    plt.show() ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_classes(df_train_train,\"TRAIN DATA\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_classes(df_train_valid,'VALIDATION DATA')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file = open('df_train_train', 'rb')\ndf_train_train = pickle.load(file)\nfile.close()\n\nfile = open('df_train_test', 'rb')\ndf_train_test = pickle.load(file)\nfile.close()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_train_train.shape,df_train_test.shape)\nprint(len(os.listdir('./train_images_resized_preprocessed')),len(os.listdir('./test_images_resized_preprocessed')))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG_SIZE  = 512","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''Function loads an image from Folder , Resizes and saves in another directory '''\n\ndef image_resize_save(file):\n    input_filepath = os.path.join('./','train_images','{}.png'.format(file))\n    output_filepath = os.path.join('./','valid_images_resized','{}.png'.format(file))\n    img = cv2.imread(input_filepath)\n    cv2.imwrite(output_filepath, cv2.resize(img, (IMG_SIZE,IMG_SIZE)))\n#image_resize_save(df_train.id_code.iloc[201])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''This Function uses Multi processing for faster saving of images into folder'''\n\ndef multiprocess_image_downloader(process:int, imgs:list):\n    \"\"\"\n    Inputs:\n        process: (int) number of process to run\n        imgs:(list) list of images\n    \"\"\"\n    print(f'MESSAGE: Running {process} process')\n    results = ThreadPool(process).map(image_resize_save, imgs)\n    return results","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"multiprocess_image_downloader(6, list(df_train_valid.id_code.values))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}