{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Medical Classification","metadata":{}},{"cell_type":"code","source":"! pip install pillow==9.2.0 scikit-image==0.19.2","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing","metadata":{}},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot\nimport numpy\nimport os\nimport pandas\nimport PIL\nimport PIL.ImageFile\nimport seaborn\nimport skimage\nimport time\n\nroot_path = '/content/drive/MyDrive/Projects/Siraj Raval/Medical Classification/medical_classification'\n\nclass EDA:\n    \"\"\"\n    Exploratory Data Analysis\n    \"\"\"\n    def __init__(self):\n        dict_labels = {\n            0: \"No DR\",\n            1: \"Mild\",\n            2: \"Moderate\",\n            3: \"Severe\",\n            4: \"Proliferative DR\"}\n\n    def __call__(self):\n        labels = pandas.read_csv(\"labels/trainLabels.csv\")\n        plot_classification_frequency(labels, \"level\", \"Retinopathy_vs_Frequency_All\")\n        plot_classification_frequency(labels, \"level\", \"Retinopathy_vs_Frequency_Binary\", True)\n\n    def change_labels(df, category):\n        \"\"\"\n        Changes the labels for a binary classification.\n        Either the person has a degree of retinopathy, or they don't.\n        Parameters\n            df: Pandas DataFrame of the image name and labels\n            category: column of the labels\n        Return\n            Column containing a binary classification of 0 or 1\n        \"\"\"\n        return [1 if l > 0 else 0 for l in df[category]]\n\n    def plot_classification_frequency(df, category, file_name, convert_labels=False):\n        \"\"\"\n        Plots the frequency at which labels occur.\n        Parameters\n            df: Pandas DataFrame of the image name and labels\n            category: category of labels, from 0 to 4\n            file_name: file name of the image\n            convert_labels: argument specified for converting to binary classification\n        Return\n            None\n        \"\"\"\n        if convert_labels == True:\n            labels['level'] = change_labels(labels, 'level')\n        seaborn.set(style=\"whitegrid\", color_codes=True)\n        seaborn.countplot(x=category, data=labels)\n        pyplot.title('Retinopathy vs Frequency')\n        pyplot.savefig(file_name)\n        return","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EDA()()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CropAndResizeImages:\n    \"\"\"\n    TODO: docstring\n    \"\"\"\n    PIL.ImageFile.LOAD_TRUNCATED_IMAGES = True\n\n    def crop_and_resize_images(path, new_path, cropx, cropy, img_size):\n        \"\"\"\n        Crops, resizes, and stores all images from a directory in a new directory.\n\n        Parameters\n            path: Path where the current, unscaled images are contained.\n            new_path: Path to save the resized images.\n            img_size: New size for the rescaled images.\n\n        Return\n            none\n        \"\"\"\n        if not os.path.exists(new_path):\n            os.makedirs(new_path)\n        dirs = [l for l in os.listdir(path)]\n        total = 0\n        for item in dirs:\n            img = skimage.io.imread(path + item)\n            y, x, channel = img.shape\n            startx = x // 2 - (cropx // 2)\n            starty = y // 2 - (cropy // 2)\n            img = img[starty:starty+cropy, startx:startx+cropx]\n            img = skimage.transform.resize(img, (img_size, img_size))\n            skimage.io.imsave(str(new_path + item), img)\n            total += 1\n            print('saving:', item, total)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CropAndResizeImages.crop_and_resize_images(\n    path=os.path.join(root_path, 'data/train/'),\n    new_path=os.path.join(root_path, 'data/train-resized-256/'),\n    cropx=1800, cropy=1800, img_size=256)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CropAndResizeImages.crop_and_resize_images(\n    path=os.path.join(root_path, 'data/test/'),\n    new_path=os.path.join(root_path, 'data/test-resized-256/'),\n    cropx=1800, cropy=1800, img_size=256)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class PreprocessImages:\n    \"\"\"\n    TODO: docstring\n    \"\"\"\n    def __call__(self):\n        \"\"\"\n        TODO: docstring\n        \"\"\"\n        trainLabels = pandas.read_csv(os.path.join(root_path, 'data/trainLabels.csv'))\n        trainLabels['image'] = [i + '.jpeg' for i in trainLabels['image']]\n        trainLabels['black'] = numpy.nan\n        trainLabels['black'] = self.find_black_images(os.path.join(\n            root_path, 'data/train-resized-256/'), trainLabels)\n        trainLabels = trainLabels.loc[trainLabels['black'] == 0]\n        trainLabels.to_csv(os.path.join(\n            root_path, 'data/trainLabels-master.csv'), index=False, header=True)\n        print('completed')\n\n    def find_black_images(self, file_path, df):\n        \"\"\"\n        Creates a column of images that are not black (numpy.mean(img) != 0)\n\n        Parameters\n            file_path: file_path to the images to be analyzed.\n            df: Pandas DataFrame that includes all labeled image names.\n            column: column in DataFrame query is evaluated against.\n\n        Return\n            Column indicating if the photo is pitch black or not.\n        \"\"\"\n        lst_imgs = [l for l in df['image']]\n        return [1 if numpy.mean(numpy.array(\n            PIL.Image.open(file_path + img))) == 0 else 0 for img in lst_imgs]\n    \n    def rename_images(self, src_dir, new_prefix):\n        \"\"\"\n        TODO: docstring\n        \"\"\"\n        for file_name in os.listdir(src_dir):\n            os.rename(\n                os.path.join(src_dir, file_name),\n                os.path.join(src_dir, new_prefix + file_name))\n            print(file_name + ' -> ' + new_prefix + file_name)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PreprocessImages()()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class RotateImages:\n    \"\"\"\n    TODO: docstring\n    \"\"\"\n    def __call__(self):\n        \"\"\"\n        TODO: docstring\n        \"\"\"\n        trainLabels = pandas.read_csv(os.path.join(\n            root_path, 'data/trainLabels-master.csv'))\n        trainLabels['image'] = trainLabels['image'].str.rstrip('.jpeg')\n        trainLabels_no_DR = trainLabels[trainLabels['level'] == 0]\n        trainLabels_DR = trainLabels[trainLabels['level'] >= 1]\n        lst_imgs_no_DR = [i for i in trainLabels_no_DR['image']]\n        lst_imgs_DR = [i for i in trainLabels_DR['image']]\n        print('mirroring non-DR images')\n        self.mirror_images(os.path.join(\n            root_path, 'data/train-resized-256/'), 1, lst_imgs_no_DR)\n        print('mirroring DR images')\n        print('rotating 90 degrees')\n        self.rotate_images(os.path.join(\n            root_path, 'data/train-resized-256/'), 90, lst_imgs_DR)\n        print('rotating 120 degrees')\n        self.rotate_images(os.path.join(\n            root_path, 'data/train-resized-256/'), 120, lst_imgs_DR)\n        print('rotating 180 degrees')\n        self.rotate_images(os.path.join(\n            root_path, 'data/train-resized-256/'), 180, lst_imgs_DR)\n        print('rotating 270 degrees')\n        self.rotate_images(os.path.join(\n            root_path, 'data/train-resized-256/'), 270, lst_imgs_DR)\n        print('mirroring DR images')\n        self.mirror_images(os.path.join(\n            root_path, 'data/train-resized-256/'), 0, lst_imgs_DR)\n        print('completed')\n\n    def mirror_images(self, file_path, mirror_direction, lst_imgs):\n        \"\"\"\n        Mirrors image left or right, based on criteria specified.\n\n        Parameters\n            file_path: file path to the folder containing images.\n            mirror_direction: criteria for mirroring left or right.\n            lst_imgs: list of image strings.\n\n        Return\n            None\n        \"\"\"\n        for l in lst_imgs:\n            img = cv2.imread(file_path + str(l) + '.jpeg')\n            img = cv2.flip(img, 1)\n            cv2.imwrite(file_path + str(l) + '_mir' + '.jpeg', img)\n\n    def rotate_images(self, file_path, degrees_of_rotation, lst_imgs):\n        \"\"\"\n        Rotates image based on a specified amount of degrees.\n\n        Parameters\n            file_path: file path to the folder containing images.\n            degrees_of_rotation: Integer, specifying degrees to rotate the\n            image. Set number from 1 to 360.\n            lst_imgs: list of image strings.\n\n        Return\n            None\n        \"\"\"\n        for l in lst_imgs:\n            img = skimage.io.imread(file_path + str(l) + '.jpeg')\n            img = skimage.transform.rotate(img, degrees_of_rotation)\n            skimage.io.imsave(file_path + str(l) + '_' + str(degrees_of_rotation) + '.jpeg', img)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RotateImages()()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ReconcileLabels:\n    \"\"\"\n    TODO: docstring\n    \"\"\"\n    def reconcile_labels():\n        \"\"\"\n        TODO: docstring\n        \"\"\"\n        trainLabels = pandas.read_csv(os.path.join(\n            root_path, 'data/trainLabels-master.csv'))\n        lst_imgs = [i for i in os.listdir(os.path.join(\n            root_path, 'data/train-resized-256/')) if i != '.DS_Store']\n        new_trainLabels = pandas.DataFrame({'image': lst_imgs})\n        new_trainLabels['image2'] = new_trainLabels.image\n        new_trainLabels['image2'] = new_trainLabels.loc[:, 'image2'].apply(\n            lambda x: '_'.join(x.split('_')[0:2]))\n        new_trainLabels['image2'] = new_trainLabels.loc[:, 'image2'].apply(\n            lambda x: '_'.join(x.split('_')[0:2]).strip('.jpeg') + '.jpeg')\n        new_trainLabels.columns = ['train_image_name', 'image']\n        trainLabels = pandas.merge(\n            trainLabels, new_trainLabels, how='outer', on='image')\n        trainLabels.drop(['black'], axis=1, inplace=True)\n        trainLabels = trainLabels.dropna()\n        print(trainLabels.shape)\n        print('writing CSV')\n        trainLabels.to_csv(os.path.join(\n            root_path, 'data/trainLabels-master_256_v2.csv'),\n            index=False, header=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ReconcileLabels.reconcile_labels()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cupy","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ImageToArray:\n    \"\"\"\n    TODO: docstring\n    \"\"\"\n    def image_to_array():\n        \"\"\"\n        TODO: docstring\n        \"\"\"\n        labels = pandas.read_csv(os.path.join(\n            root_path, 'data/trainLabels-master_256_v2.csv'))\n        labels = labels[:10000]\n        print('writing train array')\n        lst_imgs = [l for l in labels['train_image_name']]\n        X_train = cupy.array([cupy.array(PIL.Image.open(os.path.join(\n            root_path, 'data/train-resized-256/') + img)) for img in lst_imgs])\n        print(X_train.shape)\n        print('saving train array')\n        cupy.save(os.path.join(root_path, 'data/X_train_256_v2-10000.npy'), X_train)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ImageToArray.image_to_array()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training","metadata":{}},{"cell_type":"code","source":"import cupy\nimport numpy\nimport os\nimport pandas\nimport sklearn.model_selection\nimport sklearn.utils\nimport tensorflow\n\nroot_path = '/content/drive/MyDrive/Projects/Siraj Raval/Medical Classification/medical_classification'\n\nclass EyeNet:\n    \"\"\"\n    TODO: docstring\n    \"\"\"\n    def __init__(self):\n        \"\"\"\n        TODO: docstring\n        \"\"\"\n        cupy.random.seed(1337)\n        self.channels = 3\n        self.img_cols = 256\n        self.img_rows = 256\n        self.nb_classes = 5\n        self.split_data(\n            y_file_path=os.path.join(root_path, 'data/trainLabels-master_256_v2.csv'),\n            X=os.path.join(root_path, 'data/X_train_256_v2-10000.npy'))\n        self.reshape_data(self.img_rows, self.img_cols, self.channels, self.nb_classes)\n        model = self.cnn_model(\n            nb_filters=32, kernel_size=(4, 4), batch_size=512, nb_epoch=50)\n        precision, recall, f1, cohen_kappa, quad_kappa = self.predict()\n        print('Precision:', precision)\n        print('Recall:', recall)\n        print('F1:', f1)\n        print('Cohen Kappa Score:', cohen_kappa)\n        print('Quadratic Kappa:', quad_kappa)\n        self.save_model(score=recall, model_name='DR_Class')\n\n    def cnn_model(self, nb_filters, kernel_size, batch_size, nb_epoch):\n        \"\"\"\n        Define and run the Convolutional Neural Network.\n\n        Parameters\n            nb_filters: initial number of filters\n            kernel_size: initial size of kernel\n            batch_size: batch size for the model\n            nb_epoch: number of epochs\n\n        Return\n            Fitted CNN model\n        \"\"\"\n        model = tensorflow.keras.models.Sequential()\n        model.add(tensorflow.keras.layers.convolutional.Conv2D(\n            nb_filters, (kernel_size[0], kernel_size[1]), padding='valid', strides=1,\n            input_shape=(self.img_rows, self.img_cols, self.channels), activation='relu'))\n        model.add(tensorflow.keras.layers.convolutional.Conv2D(\n            nb_filters, (kernel_size[0], kernel_size[1]), activation='relu'))\n        model.add(tensorflow.keras.layers.convolutional.Conv2D(\n            nb_filters, (kernel_size[0], kernel_size[1]), activation='relu'))\n        model.add(tensorflow.keras.layers.MaxPooling2D(pool_size=(8, 8)))\n        model.add(tensorflow.keras.layers.Flatten())\n        print('model flattened out to: ', model.output_shape)\n        model.add(tensorflow.keras.layers.Dense(2048, activation='relu'))\n        model.add(tensorflow.keras.layers.Dropout(0.25))\n        model.add(tensorflow.keras.layers.Dense(2048, activation='relu'))\n        model.add(tensorflow.keras.layers.Dropout(0.25))\n        model.add(tensorflow.keras.layers.Dense(self.nb_classes, activation='softmax'))\n        #model = tensorflow.keras.utils.multi_gpu_model(model, gpus=8)\n        model.compile(\n            loss='categorical_crossentropy', optimizer='adam', metrics=['accuracy'])\n        stop = tensorflow.keras.callbacks.EarlyStopping(\n            monitor='val_acc', min_delta=0.001, patience=2, mode='auto')\n        model.fit(\n            self.X_train, self.y_train, batch_size=batch_size, epochs=nb_epoch,\n            verbose=1, validation_split=0.2, callbacks=[stop])\n        return model\n\n    def reshape_data(self, img_rows, img_cols, channels, nb_classes):\n        \"\"\"\n        Reshapes arrays into format for MXNet.\n\n        Parameters\n            img_rows: image array height\n            img_cols: image array width\n            channels: specify if image is grayscale (1) or RGB (3)\n            nb_classes: number of image classes/categories\n\n        Return\n            none\n        \"\"\"\n        self.X_train = self.X_train.reshape(\n            self.X_train.shape[0], img_rows, img_cols, channels)\n        self.X_train = self.X_train.astype('float32')\n        self.X_train /= 255\n        self.y_train = tensorflow.keras.utils.to_categorical(\n            self.y_train, self.nb_classes)\n        self.X_test = self.X_test.reshape\n        (2000, img_rows, img_cols, channels)\n        #self.X_test = self.X_test.astype('float32')\n        self.X_test /= 255\n        self.y_test = tensorflow.keras.utils.to_categorical(\n            self.y_test, self.nb_classes)\n        print('X_train shape:', self.X_train.shape)\n        print('X_test shape:', self.X_test.shape)\n        print('y_train shape:', self.y_train.shape)\n        print('y_test shape:', self.y_test.shape)\n\n    def save_model(self, score, model_name):\n        \"\"\"\n        Saves the model, based on scoring criteria input.\n\n        Parameters\n            score: scoring metric used to save model\n            model_name: name for the model to be saved\n\n        Return\n            none\n        \"\"\"\n        if score >= 0.75:\n            print('saving model')\n            self.model.save(model_name + '_recall_' + str(round(score, 4)) + '.h5')\n        else:\n            print('model not saved. score:', score)\n\n    def split_data(self, y_file_path, X, test_size=0.2):\n        \"\"\"\n        Split data into test and training data sets.\n\n        Parameters\n            y_file_path: path to CSV containing labels\n            X: NumPy array of arrays\n            test_data_size: size of test/train split. Value from 0 to 1\n        \n        Return\n            none\n        \"\"\"\n        labels = pandas.read_csv(y_file_path)\n        labels = labels[:10000]\n        X = cupy.load(X)\n        y = cupy.array(labels['level'])\n        self.weights = sklearn.utils.class_weight.compute_class_weight(\n            'balanced', classes=numpy.unique(y), y=y)\n        self.X_train, self.X_test, self.y_train, self.y_test = \\\n            sklearn.model_selection.train_test_split(\n                X, y, test_size=test_size, random_state=42)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EyeNet()","metadata":{},"execution_count":null,"outputs":[]}]}