{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"},{"sourceId":187731,"sourceType":"datasetVersion","datasetId":80814}],"dockerImageVersionId":27938,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport tensorflow as tf\nfrom matplotlib import pyplot as plt\nfrom sklearn.metrics import cohen_kappa_score\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.applications.densenet import DenseNet121\nimport keras\nimport cv2\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\nimport cv2\nimport os\nfrom keras.callbacks import Callback\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.utils.multiclass import unique_labels\nfrom sklearn.utils import class_weight\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:33:56.262679Z","iopub.execute_input":"2025-06-30T06:33:56.262948Z","iopub.status.idle":"2025-06-30T06:33:58.347124Z","shell.execute_reply.started":"2025-06-30T06:33:56.262907Z","shell.execute_reply":"2025-06-30T06:33:58.346122Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# borrowed from https://www.kaggle.com/mathormad/aptos-resnet50-baseline\nclass QWKCallback(Callback):\n    def __init__(self,validation_data):\n        super(Callback, self).__init__()\n        self.X = validation_data[0]\n        self.Y = validation_data[1]\n        self.history = []\n    def on_epoch_end(self, epoch, logs={}):\n        pred = self.model.predict(self.X)\n        score = cohen_kappa_score(np.argmax(self.Y,axis=1),np.argmax(pred,axis=1),labels=[0,1,2,3,4],weights='quadratic')\n        print(\"Epoch {} : QWK: {}\".format(epoch,score))\n        self.history.append(score)\n        if score >= max(self.history):\n            print('saving checkpoint: ', score)\n            self.model.save('../working/Resnet50_bestqwk.h5')\n        \n        \n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:33:58.348754Z","iopub.execute_input":"2025-06-30T06:33:58.348971Z","iopub.status.idle":"2025-06-30T06:33:58.355647Z","shell.execute_reply.started":"2025-06-30T06:33:58.348934Z","shell.execute_reply":"2025-06-30T06:33:58.354700Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# borrowed from https://github.com/yu4u/mixup-generator\nclass MixupGenerator():\n    def __init__(self, X_train, y_train, batch_size=32, alpha=0.2, shuffle=True, datagen=None):\n        self.X_train = X_train\n        self.y_train = y_train\n        self.batch_size = batch_size\n        self.alpha = alpha\n        self.shuffle = shuffle\n        self.sample_num = len(X_train)\n        self.datagen = datagen\n\n    def __call__(self):\n        while True:\n            indexes = self.__get_exploration_order()\n            itr_num = int(len(indexes) // (self.batch_size * 2))\n\n            for i in range(itr_num):\n                batch_ids = indexes[i * self.batch_size * 2:(i + 1) * self.batch_size * 2]\n                X, y = self.__data_generation(batch_ids)\n\n                yield X, y\n\n    def __get_exploration_order(self):\n        indexes = np.arange(self.sample_num)\n\n        if self.shuffle:\n            np.random.shuffle(indexes)\n\n        return indexes\n\n    def __data_generation(self, batch_ids):\n        _, h, w, c = self.X_train.shape\n        l = np.random.beta(self.alpha, self.alpha, self.batch_size)\n        X_l = l.reshape(self.batch_size, 1, 1, 1)\n        y_l = l.reshape(self.batch_size, 1)\n\n        X1 = self.X_train[batch_ids[:self.batch_size]]\n        X2 = self.X_train[batch_ids[self.batch_size:]]\n        X = X1 * X_l + X2 * (1 - X_l)\n\n        if self.datagen:\n            for i in range(self.batch_size):\n                X[i] = self.datagen.random_transform(X[i])\n                X[i] = self.datagen.standardize(X[i])\n\n        if isinstance(self.y_train, list):\n            y = []\n\n            for y_train_ in self.y_train:\n                y1 = y_train_[batch_ids[:self.batch_size]]\n                y2 = y_train_[batch_ids[self.batch_size:]]\n                y.append(y1 * y_l + y2 * (1 - y_l))\n        else:\n            y1 = self.y_train[batch_ids[:self.batch_size]]\n            y2 = self.y_train[batch_ids[self.batch_size:]]\n            y = y1 * y_l + y2 * (1 - y_l)\n\n        return X, y","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:33:58.356795Z","iopub.execute_input":"2025-06-30T06:33:58.357041Z","iopub.status.idle":"2025-06-30T06:33:58.373547Z","shell.execute_reply.started":"2025-06-30T06:33:58.356995Z","shell.execute_reply":"2025-06-30T06:33:58.372949Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# borrowed from scikit learn\ndef plot_confusion_matrix(y_true, y_pred, classes,\n                          normalize=False,\n                          title=None,\n                          cmap=plt.cm.Blues):\n    \"\"\"\n    This function prints and plots the confusion matrix.\n    Normalization can be applied by setting `normalize=True`.\n    \"\"\"\n    if not title:\n        if normalize:\n            title = 'Normalized confusion matrix'\n        else:\n            title = 'Confusion matrix, without normalization'\n\n    # Compute confusion matrix\n    cm = confusion_matrix(y_true, y_pred)\n    # Only use the labels that appear in the data\n    classes = classes[unique_labels(y_true, y_pred)]\n    if normalize:\n        cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n        print(\"Normalized confusion matrix\")\n    else:\n        print('Confusion matrix, without normalization')\n\n    print(cm)\n\n    fig, ax = plt.subplots()\n    im = ax.imshow(cm, interpolation='nearest', cmap=cmap)\n    ax.figure.colorbar(im, ax=ax)\n    # We want to show all ticks...\n    ax.set(xticks=np.arange(cm.shape[1]),\n           yticks=np.arange(cm.shape[0]),\n           # ... and label them with the respective list entries\n           xticklabels=classes, yticklabels=classes,\n           title=title,\n           ylabel='True label',\n           xlabel='Predicted label')\n\n    # Rotate the tick labels and set their alignment.\n    plt.setp(ax.get_xticklabels(), rotation=45, ha=\"right\",\n             rotation_mode=\"anchor\")\n\n    # Loop over data dimensions and create text annotations.\n    fmt = '.2f' if normalize else 'd'\n    thresh = cm.max() / 2.\n    for i in range(cm.shape[0]):\n        for j in range(cm.shape[1]):\n            ax.text(j, i, format(cm[i, j], fmt),\n                    ha=\"center\", va=\"center\",\n                    color=\"white\" if cm[i, j] > thresh else \"black\")\n    fig.tight_layout()\n    return ax\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:33:58.374835Z","iopub.execute_input":"2025-06-30T06:33:58.375118Z","iopub.status.idle":"2025-06-30T06:33:58.388496Z","shell.execute_reply.started":"2025-06-30T06:33:58.375073Z","shell.execute_reply":"2025-06-30T06:33:58.387814Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_raw_images_df(data_frame,filenamecol,labelcol,img_size,n_classes):\n    n_images = len(data_frame)\n    X = np.empty((n_images,img_size,img_size,3))\n    Y = np.zeros((n_images,n_classes))\n    for index,entry in data_frame.iterrows():\n        Y[index,entry[labelcol]] = 1 # one hot encoding of the label\n        # Load the image and resize\n        img = cv2.imread(entry[filenamecol])\n        X[index,:] = cv2.resize(img, (img_size, img_size))\n        X[index,:] = X[index,:] / 255.0\n    return X,Y","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:33:58.391942Z","iopub.execute_input":"2025-06-30T06:33:58.392131Z","iopub.status.idle":"2025-06-30T06:33:58.402113Z","shell.execute_reply.started":"2025-06-30T06:33:58.392098Z","shell.execute_reply":"2025-06-30T06:33:58.401513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"batch_size = 32\nimg_size = 224","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:33:58.404996Z","iopub.execute_input":"2025-06-30T06:33:58.405208Z","iopub.status.idle":"2025-06-30T06:33:58.416896Z","shell.execute_reply.started":"2025-06-30T06:33:58.405170Z","shell.execute_reply":"2025-06-30T06:33:58.416171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_raw_data = pd.read_csv(\"../input/aptos2019-blindness-detection/train.csv\")\ntrain_raw_data[\"filename\"] = train_raw_data[\"id_code\"].map(lambda x:os.path.join(\"../input/aptos2019-blindness-detection/train_images\",x+\".png\"))\ntrain_raw_data.diagnosis.hist() # See the distribution of the classes\n# train_raw_data.dtypes\n\n# # train_data[\"diagnosis\"] = train_data[\"diagnosis\"].astype(str)\n# # print(train_data.head())\n# # print(train_data.diagnosis.unique()) # Look at different types of classes\n# # labels = list(map(str,range(5)))\n# # print(labels)","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:33:58.418387Z","iopub.execute_input":"2025-06-30T06:33:58.418690Z","iopub.status.idle":"2025-06-30T06:33:58.684500Z","shell.execute_reply.started":"2025-06-30T06:33:58.418637Z","shell.execute_reply":"2025-06-30T06:33:58.683754Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_title = {\"0\" : \"No DR\",\"1\" : \"Mild\",\"2\" : \"Moderate\",\"3\" :\"Severe\",\"4\" : \"Proliferative DR\"}\nclass_labels=[\"No DR\",\"Mild\",\"Moderate\",\"Severe\",\"Proliferative DR\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:33:58.686136Z","iopub.execute_input":"2025-06-30T06:33:58.686451Z","iopub.status.idle":"2025-06-30T06:33:58.691028Z","shell.execute_reply.started":"2025-06-30T06:33:58.686401Z","shell.execute_reply":"2025-06-30T06:33:58.690155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Display some images\nfigure, ax = plt.subplots(5,2)\nax = ax.flatten()\nfor i,row in train_raw_data.iloc[0:10,:].iterrows():\n    ax[i].imshow(cv2.imread(os.path.join(\"../input/aptos2019-blindness-detection/train_images\",row[\"id_code\"]+\".png\")))\n    ax[i].set_title(label_title[str(row[\"diagnosis\"])])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:33:58.692100Z","iopub.execute_input":"2025-06-30T06:33:58.692395Z","iopub.status.idle":"2025-06-30T06:34:04.078061Z","shell.execute_reply.started":"2025-06-30T06:33:58.692347Z","shell.execute_reply":"2025-06-30T06:34:04.077223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df,val_df = train_test_split(train_raw_data,random_state=42,shuffle=True,test_size=0.333)\ntrain_df.reset_index(drop=True,inplace=True)\nval_df.reset_index(drop=True,inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:34:04.079308Z","iopub.execute_input":"2025-06-30T06:34:04.079742Z","iopub.status.idle":"2025-06-30T06:34:04.092162Z","shell.execute_reply.started":"2025-06-30T06:34:04.079693Z","shell.execute_reply":"2025-06-30T06:34:04.091448Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train,Y_train = load_raw_images_df(train_df,\"filename\",\"diagnosis\",img_size,5)\nX_val,Y_val = load_raw_images_df(val_df,\"filename\",\"diagnosis\",img_size,5)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:34:04.093856Z","iopub.execute_input":"2025-06-30T06:34:04.094139Z","iopub.status.idle":"2025-06-30T06:40:04.240683Z","shell.execute_reply.started":"2025-06-30T06:34:04.094092Z","shell.execute_reply":"2025-06-30T06:40:04.239761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Y_train_labels = np.argmax(Y_train,axis=1)\nclass_weights = class_weight.compute_class_weight('balanced',np.unique(Y_train_labels),Y_train_labels)\ncls_wt_dict = dict(enumerate(class_weights))\nprint(cls_wt_dict)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:40:04.242115Z","iopub.execute_input":"2025-06-30T06:40:04.242331Z","iopub.status.idle":"2025-06-30T06:40:04.248387Z","shell.execute_reply.started":"2025-06-30T06:40:04.242295Z","shell.execute_reply":"2025-06-30T06:40:04.247725Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"datagen = ImageDataGenerator(\n            \n            zoom_range=0.15,  # set range for random zoom\n        # set mode for filling points outside the input boundaries\n        fill_mode='constant',\n        cval=0.,  # value used for fill_mode = \"constant\"\n        horizontal_flip=True,  # randomly flip images\n        vertical_flip=True,  # randomly flip images\n)\ntraining_generator = MixupGenerator(X_train, Y_train, batch_size=batch_size, alpha=0.2, datagen=datagen)()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:40:04.249858Z","iopub.execute_input":"2025-06-30T06:40:04.250129Z","iopub.status.idle":"2025-06-30T06:40:04.261095Z","shell.execute_reply.started":"2025-06-30T06:40:04.250079Z","shell.execute_reply":"2025-06-30T06:40:04.260498Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def buildModel():\n    DenseNet121_model = DenseNet121(include_top=False,weights=None,input_tensor=keras.layers.Input(shape=(img_size,img_size,3)))\n    DenseNet121_model.load_weights('../input/densenet-keras/DenseNet-BC-121-32-no-top.h5')\n#     model = keras.Sequential()\n    \n#     model.add(keras.layers.Conv2D(filters = 32, kernel_size = (5,5),padding = 'same',activation ='relu', \n#                       input_shape = (img_size,img_size,3)))\n#     model.add(keras.layers.MaxPooling2D(pool_size=(2,2)))\n    \n#     model.add(keras.layers.Conv2D(filters = 64, kernel_size = (3,3),padding = 'Same',activation ='relu'))\n#     model.add(keras.layers.MaxPooling2D(pool_size=(2,2), strides=(2,2)))\n    \n#     model.add(keras.layers.Conv2D(filters =96, kernel_size = (3,3),padding = 'Same',activation ='relu'))\n#     model.add(keras.layers.MaxPooling2D(pool_size=(2,2), strides=(2,2)))\n\n#     model.add(keras.layers.Conv2D(filters = 96, kernel_size = (3,3),padding = 'Same',activation ='relu'))\n#     model.add(keras.layers.MaxPooling2D(pool_size=(2,2), strides=(2,2)))\n    \n#     model.add(keras.layers.Flatten())\n#     model.add(keras.layers.Dense(units = 512, activation = 'relu'))\n#     model.add(keras.layers.Dense(units = 5, activation = 'softmax'))\n    \n    p  = keras.layers.GlobalAveragePooling2D()(DenseNet121_model.output)\n#     fl = keras.layers.Flatten()(p)\n#     d2 = keras.layers.Dense(units = 1024, activation = 'relu',kernel_regularizer= keras.regularizers.l2(0.001))(p)\n#     d1 = keras.layers.Dense(units = 512, activation = 'relu',kernel_regularizer= keras.regularizers.l2(0.001))(d2)\n    d11 = keras.layers.Dense(units = 256, activation = 'relu',kernel_regularizer= keras.regularizers.l2(0.0001))(p)\n    o1 = keras.layers.Dense(units = 5, activation = 'softmax')(d11)\n    model = keras.models.Model(inputs = DenseNet121_model.input,outputs = o1)\n    sgd = keras.optimizers.SGD(lr=0.01, decay=1e-6, momentum=0.9, nesterov=True)\n    model.compile(optimizer=sgd,loss='categorical_crossentropy', metrics = ['accuracy'])\n    print(model.summary())\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:40:04.262345Z","iopub.execute_input":"2025-06-30T06:40:04.262639Z","iopub.status.idle":"2025-06-30T06:40:04.270545Z","shell.execute_reply.started":"2025-06-30T06:40:04.262575Z","shell.execute_reply":"2025-06-30T06:40:04.269880Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mymodel = buildModel()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:40:04.271810Z","iopub.execute_input":"2025-06-30T06:40:04.272087Z","iopub.status.idle":"2025-06-30T06:40:24.546729Z","shell.execute_reply.started":"2025-06-30T06:40:04.272035Z","shell.execute_reply":"2025-06-30T06:40:24.546038Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"EPOCHS = 50\nearlystop = keras.callbacks.EarlyStopping(patience=10)\nlearning_rate_reduction = keras.callbacks.ReduceLROnPlateau(monitor='val_acc', \n                                            patience=2, \n                                            verbose=1, \n                                            factor=0.5, \n                                            min_lr=0.00001)\ncheckpoint = keras.callbacks.ModelCheckpoint('../working/DenseNet121.h5', monitor='val_loss', verbose=1, \n                             save_best_only=True, mode='min', save_weights_only = True)\nqwk = QWKCallback((X_val,Y_val))\nmycallbacks = [earlystop, learning_rate_reduction,checkpoint,qwk]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:40:24.547824Z","iopub.execute_input":"2025-06-30T06:40:24.548063Z","iopub.status.idle":"2025-06-30T06:40:24.557518Z","shell.execute_reply.started":"2025-06-30T06:40:24.548015Z","shell.execute_reply":"2025-06-30T06:40:24.556811Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(qwk)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:40:24.558847Z","iopub.execute_input":"2025-06-30T06:40:24.559075Z","iopub.status.idle":"2025-06-30T06:40:24.571961Z","shell.execute_reply.started":"2025-06-30T06:40:24.559029Z","shell.execute_reply":"2025-06-30T06:40:24.571043Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Warm up the model with class weights\nEPOCHS = 10\nhistory = mymodel.fit_generator(training_generator,steps_per_epoch = X_train.shape[0] // batch_size,epochs = EPOCHS,\n                         validation_data = (X_val,Y_val),\n                         validation_steps = 10,\n                         workers = 2,use_multiprocessing=True,\n                         verbose=2, callbacks=mycallbacks,\n                         class_weight=cls_wt_dict)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:40:24.573023Z","iopub.execute_input":"2025-06-30T06:40:24.573207Z","iopub.status.idle":"2025-06-30T06:48:15.242806Z","shell.execute_reply.started":"2025-06-30T06:40:24.573175Z","shell.execute_reply":"2025-06-30T06:48:15.241581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"EPOCHS = 50\nhistory = mymodel.fit_generator(training_generator,steps_per_epoch = X_train.shape[0] // batch_size,epochs = EPOCHS,\n                         validation_data = (X_val,Y_val),\n                         validation_steps = 10,\n                         workers = 2,use_multiprocessing=True,\n                         verbose=2, callbacks=mycallbacks)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:48:15.244918Z","iopub.execute_input":"2025-06-30T06:48:15.245185Z","iopub.status.idle":"2025-06-30T06:55:02.419849Z","shell.execute_reply.started":"2025-06-30T06:48:15.245126Z","shell.execute_reply":"2025-06-30T06:55:02.418836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mymodel.save_weights(\"model.h5\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:55:02.422459Z","iopub.execute_input":"2025-06-30T06:55:02.422821Z","iopub.status.idle":"2025-06-30T06:55:02.981341Z","shell.execute_reply.started":"2025-06-30T06:55:02.422755Z","shell.execute_reply":"2025-06-30T06:55:02.980477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Y_val_pred = mymodel.predict_on_batch(X_val)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:55:02.983009Z","iopub.execute_input":"2025-06-30T06:55:02.983306Z","iopub.status.idle":"2025-06-30T06:55:11.907096Z","shell.execute_reply.started":"2025-06-30T06:55:02.983250Z","shell.execute_reply":"2025-06-30T06:55:11.906420Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Y_val_pred_hot = np.argmax(Y_val_pred,axis=1)\nY_val_actual_hot = np.argmax(Y_val,axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:55:11.908472Z","iopub.execute_input":"2025-06-30T06:55:11.908787Z","iopub.status.idle":"2025-06-30T06:55:11.913307Z","shell.execute_reply.started":"2025-06-30T06:55:11.908725Z","shell.execute_reply":"2025-06-30T06:55:11.912502Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, accuracy_score\nplot_confusion_matrix(Y_val_actual_hot, Y_val_pred_hot, np.array(class_labels))\naccuracy = accuracy_score(Y_val_actual_hot, Y_val_pred_hot)\nprint(f\"Validation Accuracy: {accuracy:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T07:10:23.674683Z","iopub.execute_input":"2025-06-30T07:10:23.674942Z","iopub.status.idle":"2025-06-30T07:10:24.116721Z","shell.execute_reply.started":"2025-06-30T07:10:23.674910Z","shell.execute_reply":"2025-06-30T07:10:24.115780Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# history = mymodel.fit_generator(train_gen,steps_per_epoch = 10,epochs = EPOCHS,\n#                          validation_data = val_gen,\n#                          validation_steps = 10,\n#                          workers = 2,use_multiprocessing=True,\n#                      verbose=2, callbacks=mycallbacks)\n# mymodel.save_weights(\"model.h5\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:55:12.381065Z","iopub.execute_input":"2025-06-30T06:55:12.381431Z","iopub.status.idle":"2025-06-30T06:55:12.390062Z","shell.execute_reply.started":"2025-06-30T06:55:12.381376Z","shell.execute_reply":"2025-06-30T06:55:12.388970Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train_gen = img_gen.flow_from_dataframe(\n#     dataframe=train_data,\n#     directory=\"../input/aptos2019-blindness-detection/train_images\",\n#     x_col=\"filename\",\n#     y_col=\"diagnosis\",\n#     batch_size=batch_size,\n#     shuffle=True,\n#     class_mode=\"categorical\",\n#     classes=labels,\n#     target_size=(img_size,img_size),\n#     subset='training')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:55:12.392077Z","iopub.execute_input":"2025-06-30T06:55:12.392924Z","iopub.status.idle":"2025-06-30T06:55:12.402535Z","shell.execute_reply.started":"2025-06-30T06:55:12.392868Z","shell.execute_reply":"2025-06-30T06:55:12.401578Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# val_gen = img_gen.flow_from_dataframe(\n#     dataframe=train_data,\n#     directory=\"../input/aptos2019-blindness-detection/train_images\",\n#     x_col=\"filename\",\n#     y_col=\"diagnosis\",\n#     batch_size=batch_size,\n#     shuffle=True,\n#     class_mode=\"categorical\",\n#     classes=labels,\n#     target_size=(img_size,img_size),\n#     subset='validation'\n# )\n# # dir(val_gen)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:55:12.403910Z","iopub.execute_input":"2025-06-30T06:55:12.404256Z","iopub.status.idle":"2025-06-30T06:55:12.414223Z","shell.execute_reply.started":"2025-06-30T06:55:12.404192Z","shell.execute_reply":"2025-06-30T06:55:12.413132Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# class QWK(keras.callbacks.Callback):\n#     def on_train_begin(self, logs={}):\n#         self.val_kappas = []\n\n#     def on_epoch_end(self, epoch, logs={}):\n#         print(self.__dict__)\n#         X_val, y_val = self.model.validation_data[:2]\n#         y_pred = self.model.predict(X_val)\n\n#         _val_kappa = cohen_kappa_score(\n#             y_val.argmax(axis=1), \n#             y_pred.argmax(axis=1), \n#             weights='quadratic'\n#         )\n\n#         self.val_kappas.append(_val_kappa)\n\n#         print(f\"epoch: {epoch} \\t val_kappa: {_val_kappa:.4f}\")\n\n#         return\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:55:12.416221Z","iopub.execute_input":"2025-06-30T06:55:12.416748Z","iopub.status.idle":"2025-06-30T06:55:12.427145Z","shell.execute_reply.started":"2025-06-30T06:55:12.416687Z","shell.execute_reply":"2025-06-30T06:55:12.426206Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# val_gen = img_gen.flow_from_dataframe(\n#     dataframe=train_data,\n#     directory=\"../input/aptos2019-blindness-detection/train_images\",\n#     x_col=\"filename\",\n#     y_col=\"diagnosis\",\n#     batch_size=1000,\n#     shuffle=True,\n#     class_mode=\"categorical\",\n#     classes=labels,\n#     target_size=(img_size,img_size),\n#     subset='validation'\n# )\n# # print(val_gen.batch_index)\n# # for i in range(val_gen.batch_index):\n# batch_data, batch_labels = val_gen.next()\n# batch_labels_arg = np.argmax(batch_labels,axis=1)\n# # print(batch_data.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:55:12.428486Z","iopub.execute_input":"2025-06-30T06:55:12.430024Z","iopub.status.idle":"2025-06-30T06:55:12.437185Z","shell.execute_reply.started":"2025-06-30T06:55:12.429960Z","shell.execute_reply":"2025-06-30T06:55:12.436227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# cur_batch_size = batch_data.shape[0] \n# preds = np.empty((cur_batch_size,5))\n# preds_one_hot = np.zeros((cur_batch_size,5))\n# for i in range(cur_batch_size):\n#     preds[i,:] = mymodel.predict(batch_data[i,:,:,:][np.newaxis,:])\n# args_max = np.argmax(preds,axis=1)\n# preds_one_hot[np.arange(cur_batch_size),args_max] = 1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:55:12.439345Z","iopub.execute_input":"2025-06-30T06:55:12.439932Z","iopub.status.idle":"2025-06-30T06:55:12.446220Z","shell.execute_reply.started":"2025-06-30T06:55:12.439710Z","shell.execute_reply":"2025-06-30T06:55:12.445261Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot_confusion_matrix(batch_labels_arg, args_max, np.array(class_labels))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T07:20:00.903167Z","iopub.execute_input":"2025-06-30T07:20:00.903533Z","iopub.status.idle":"2025-06-30T07:20:00.907482Z","shell.execute_reply.started":"2025-06-30T07:20:00.903454Z","shell.execute_reply":"2025-06-30T07:20:00.906454Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(2, 1, figsize=(12, 12))\nax1.plot(history.history['loss'], color='b', label=\"Training loss\")\nax1.plot(history.history['val_loss'], color='r', label=\"validation loss\")\n# ax1.set_xticks(np.arange(1, epochs, 1))\n# ax1.set_yticks(np.arange(0, 1, 0.1))\n\nax2.plot(history.history['acc'], color='b', label=\"Training accuracy\")\nax2.plot(history.history['val_acc'], color='r',label=\"Validation accuracy\")\n# ax2.set_xticks(np.arange(1, epochs, 1))\n\nlegend = plt.legend(loc='best', shadow=True)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T06:55:12.461138Z","iopub.execute_input":"2025-06-30T06:55:12.461669Z","iopub.status.idle":"2025-06-30T06:55:13.031226Z","shell.execute_reply.started":"2025-06-30T06:55:12.461593Z","shell.execute_reply":"2025-06-30T06:55:13.030491Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\n# Load metadata\ntrain_raw_data = pd.read_csv(\"../input/aptos2019-blindness-detection/train.csv\")\ntrain_raw_data[\"filename\"] = train_raw_data[\"id_code\"].map(lambda x: os.path.join(\n    \"../input/aptos2019-blindness-detection/train_images\", x + \".png\"))\n\n# Visualize class distribution\ntrain_raw_data.diagnosis.hist()\nplt.title(\"Class Distribution\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T10:21:03.708823Z","iopub.execute_input":"2025-06-30T10:21:03.709083Z","iopub.status.idle":"2025-06-30T10:21:04.008482Z","shell.execute_reply.started":"2025-06-30T10:21:03.709043Z","shell.execute_reply":"2025-06-30T10:21:04.007336Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\nfrom tqdm import tqdm\n\ndef preprocess_image(img_path, target_size=224):\n    img = cv2.imread(img_path)\n    gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n\n    # Apply Gaussian blur for background estimation\n    bg = cv2.medianBlur(gray, 51)\n    shade_corrected = cv2.subtract(gray, bg)\n    shade_corrected = cv2.normalize(shade_corrected, None, 0, 255, cv2.NORM_MINMAX)\n\n    # Resize to VGG input size\n    resized = cv2.resize(shade_corrected, (target_size, target_size))\n    return resized\n\n# Preprocess all images\nX = []\ny = []\nfor i, row in tqdm(train_raw_data.iterrows(), total=len(train_raw_data)):\n    img = preprocess_image(row[\"filename\"])\n    X.append(img)\n    y.append(row[\"diagnosis\"])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T10:21:13.837138Z","iopub.execute_input":"2025-06-30T10:21:13.837426Z","iopub.status.idle":"2025-06-30T10:31:12.269958Z","shell.execute_reply.started":"2025-06-30T10:21:13.837386Z","shell.execute_reply":"2025-06-30T10:31:12.269132Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torchvision.models as models\nimport torchvision.transforms as transforms\n\n# Load pretrained VGG-19\nvgg19 = models.vgg19(pretrained=False).features.eval()\n\n# Freeze VGG\nfor param in vgg19.parameters():\n    param.requires_grad = False\n\n# Define preprocessing transform to simulate RGB input\ntransform = transforms.Compose([\n    transforms.ToPILImage(),\n    transforms.Grayscale(num_output_channels=3),\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406],\n                         std=[0.229, 0.224, 0.225])\n])\n\ndef extract_features(images):\n    features = []\n    for img in tqdm(images):\n        tensor = transform(img).unsqueeze(0)\n        with torch.no_grad():\n            feat = vgg19(tensor)\n            features.append(feat.view(-1).numpy())  # flatten\n    return np.array(features)\n\nX_vgg = extract_features(X)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T10:37:40.382263Z","iopub.execute_input":"2025-06-30T10:37:40.382527Z","iopub.status.idle":"2025-06-30T10:52:03.683099Z","shell.execute_reply.started":"2025-06-30T10:37:40.382489Z","shell.execute_reply":"2025-06-30T10:52:03.682291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.decomposition import PCA, TruncatedSVD\n\n# Reduce dimensionality\npca = PCA(n_components=256, random_state=42)\nX_pca = pca.fit_transform(X_vgg)\n\n# Alternatively, use SVD (also called LSA/LSI)\nsvd = TruncatedSVD(n_components=256, random_state=42)\nX_svd = svd.fit_transform(X_vgg)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T10:52:03.687334Z","iopub.execute_input":"2025-06-30T10:52:03.687615Z","iopub.status.idle":"2025-06-30T10:52:23.256767Z","shell.execute_reply.started":"2025-06-30T10:52:03.687562Z","shell.execute_reply":"2025-06-30T10:52:23.255793Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import cross_val_score\n\n# Try both PCA and SVD features\nclf_pca = LogisticRegression(max_iter=1000, multi_class='multinomial', solver='lbfgs')\nclf_svd = LogisticRegression(max_iter=1000, multi_class='multinomial', solver='lbfgs')\n\nacc_pca = cross_val_score(clf_pca, X_pca, y, cv=5, scoring='accuracy')\nacc_svd = cross_val_score(clf_svd, X_svd, y, cv=5, scoring='accuracy')\n\nprint(\"PCA Accuracy:\", acc_pca.mean())\nprint(\"SVD Accuracy:\", acc_svd.mean())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T10:59:01.968315Z","iopub.execute_input":"2025-06-30T10:59:01.968631Z","iopub.status.idle":"2025-06-30T10:59:24.635340Z","shell.execute_reply.started":"2025-06-30T10:59:01.968588Z","shell.execute_reply":"2025-06-30T10:59:24.633960Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nfrom sklearn.model_selection import train_test_split\n\nX_train, X_test, y_train, y_test = train_test_split(X_svd, y, test_size=0.2, random_state=42)\nclf = LogisticRegression(max_iter=1000, multi_class='multinomial', solver='lbfgs')\nclf.fit(X_train, y_train)\n\ny_pred = clf.predict(X_test)\nprint(classification_report(y_test, y_pred))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T10:59:24.636745Z","iopub.execute_input":"2025-06-30T10:59:24.637056Z","iopub.status.idle":"2025-06-30T10:59:28.789282Z","shell.execute_reply.started":"2025-06-30T10:59:24.636989Z","shell.execute_reply":"2025-06-30T10:59:28.787661Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}