{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"train_new.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-07T02:34:44.584239Z","iopub.execute_input":"2023-11-07T02:34:44.584505Z","iopub.status.idle":"2023-11-07T02:34:44.594468Z","shell.execute_reply.started":"2023-11-07T02:34:44.584481Z","shell.execute_reply":"2023-11-07T02:34:44.593182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = EyeData(data   = test, \n                                 directory = '../input/aptos2019-blindness-detection/test_images',\n                                 transform = valid_trans,\n                                 itype = '.png')","metadata":{"execution":{"iopub.status.busy":"2023-11-07T02:34:44.595772Z","iopub.execute_input":"2023-11-07T02:34:44.596143Z","iopub.status.idle":"2023-11-07T02:34:44.602739Z","shell.execute_reply.started":"2023-11-07T02:34:44.596113Z","shell.execute_reply":"2023-11-07T02:34:44.601792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.utils import shuffle\ntrain_new = shuffle(train_new)","metadata":{"execution":{"iopub.status.busy":"2023-11-07T02:34:44.604066Z","iopub.execute_input":"2023-11-07T02:34:44.604391Z","iopub.status.idle":"2023-11-07T02:34:44.614503Z","shell.execute_reply.started":"2023-11-07T02:34:44.604361Z","shell.execute_reply":"2023-11-07T02:34:44.613604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_new = train_new.reset_index()\ntrain_new.drop(columns=['index'],inplace=True)\ntrain_new.to_csv('train_new.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-07T02:34:44.615786Z","iopub.execute_input":"2023-11-07T02:34:44.616271Z","iopub.status.idle":"2023-11-07T02:34:44.646981Z","shell.execute_reply.started":"2023-11-07T02:34:44.616195Z","shell.execute_reply":"2023-11-07T02:34:44.646241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"######################resnet50","metadata":{"execution":{"iopub.status.busy":"2023-11-07T02:34:44.648086Z","iopub.execute_input":"2023-11-07T02:34:44.648416Z","iopub.status.idle":"2023-11-07T02:34:44.652911Z","shell.execute_reply.started":"2023-11-07T02:34:44.648384Z","shell.execute_reply":"2023-11-07T02:34:44.65191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport random\nimport warnings\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import confusion_matrix, cohen_kappa_score\nfrom keras.models import Model\nfrom keras import optimizers, applications\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.callbacks import EarlyStopping, ReduceLROnPlateau\nfrom keras.layers import Dense, Dropout, GlobalAveragePooling2D, Input\n\n\nimport tensorflow as tf\n\ndef seed_everything(seed=0):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)  # Set random seed for TensorFlow\n\nseed_everything()\n\n\n%matplotlib inline\nsns.set(style=\"whitegrid\")\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2023-11-07T02:34:44.654147Z","iopub.execute_input":"2023-11-07T02:34:44.654883Z","iopub.status.idle":"2023-11-07T02:34:51.895066Z","shell.execute_reply.started":"2023-11-07T02:34:44.654827Z","shell.execute_reply":"2023-11-07T02:34:51.894235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model parameters\nBATCH_SIZE = 8\nEPOCHS = 20\nWARMUP_EPOCHS = 2\nLEARNING_RATE = 1e-4\nWARMUP_LEARNING_RATE = 1e-3\nHEIGHT = 512\nWIDTH = 512\nCANAL = 3\nN_CLASSES = train_new['diagnosis'].nunique()\nES_PATIENCE = 5\nRLROP_PATIENCE = 3\nDECAY_DROP = 0.5","metadata":{"execution":{"iopub.status.busy":"2023-11-07T02:34:51.89623Z","iopub.execute_input":"2023-11-07T02:34:51.896779Z","iopub.status.idle":"2023-11-07T02:34:51.903405Z","shell.execute_reply.started":"2023-11-07T02:34:51.896751Z","shell.execute_reply":"2023-11-07T02:34:51.902562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_new[\"id_code\"] = train_new[\"id_code\"].apply(lambda x: x if x.endswith(\".png\") else x + \".png\")\ntest_dataset.data[\"id_code\"] = test_dataset.data[\"id_code\"].apply(lambda x: x if x.endswith(\".png\") else x + \".png\")\ntrain_new['diagnosis'] = train_new['diagnosis'].astype(str)\ntrain_new.head()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-11-07T02:34:51.904502Z","iopub.execute_input":"2023-11-07T02:34:51.904768Z","iopub.status.idle":"2023-11-07T02:34:51.931937Z","shell.execute_reply.started":"2023-11-07T02:34:51.904744Z","shell.execute_reply":"2023-11-07T02:34:51.93109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.shape,train_new.shape)\nprint(test.shape,test_dataset.data.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-07T02:34:51.932999Z","iopub.execute_input":"2023-11-07T02:34:51.933284Z","iopub.status.idle":"2023-11-07T02:34:51.938111Z","shell.execute_reply.started":"2023-11-07T02:34:51.933258Z","shell.execute_reply":"2023-11-07T02:34:51.937199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset.data.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-07T02:34:51.939449Z","iopub.execute_input":"2023-11-07T02:34:51.939824Z","iopub.status.idle":"2023-11-07T02:34:51.951693Z","shell.execute_reply.started":"2023-11-07T02:34:51.939793Z","shell.execute_reply":"2023-11-07T02:34:51.950607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datagen=ImageDataGenerator(rescale=1./255, \n                                 validation_split=0.2,\n                                 horizontal_flip=True)\n\ntrain_generator=train_datagen.flow_from_dataframe(\n    dataframe=train_new,\n    directory=\"../input/aptos2019-blindness-detection/train_images/\",\n    x_col=\"id_code\",\n    y_col=\"diagnosis\",\n    batch_size=BATCH_SIZE,\n    class_mode=\"categorical\",\n    target_size=(HEIGHT, WIDTH),\n    subset='training')","metadata":{"execution":{"iopub.status.busy":"2023-11-07T02:34:51.95299Z","iopub.execute_input":"2023-11-07T02:34:51.95325Z","iopub.status.idle":"2023-11-07T02:34:53.691644Z","shell.execute_reply.started":"2023-11-07T02:34:51.953227Z","shell.execute_reply":"2023-11-07T02:34:53.690731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_generator=train_datagen.flow_from_dataframe(\n    dataframe=train_new,\n    directory=\"../input/aptos2019-blindness-detection/train_images/\",\n    x_col=\"id_code\",\n    y_col=\"diagnosis\",\n    batch_size=BATCH_SIZE,\n    class_mode=\"categorical\",    \n    target_size=(HEIGHT, WIDTH),\n    subset='validation')","metadata":{"execution":{"iopub.status.busy":"2023-11-07T02:34:53.692684Z","iopub.execute_input":"2023-11-07T02:34:53.692998Z","iopub.status.idle":"2023-11-07T02:34:53.799684Z","shell.execute_reply.started":"2023-11-07T02:34:53.692974Z","shell.execute_reply":"2023-11-07T02:34:53.798762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_datagen = ImageDataGenerator(rescale=1./255)\n\ntest_generator = test_datagen.flow_from_dataframe(  \n        dataframe=test,\n        directory = \"../input/aptos2019-blindness-detection/test_images/\",\n        x_col=\"id_code\",\n        target_size=(HEIGHT, WIDTH),\n        batch_size=1,\n        shuffle=False,\n        class_mode=None)","metadata":{"execution":{"iopub.status.busy":"2023-11-07T02:34:53.800934Z","iopub.execute_input":"2023-11-07T02:34:53.801291Z","iopub.status.idle":"2023-11-07T02:34:59.817776Z","shell.execute_reply.started":"2023-11-07T02:34:53.801259Z","shell.execute_reply":"2023-11-07T02:34:59.816822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Initialize lists to store the image dimensions\nwidths = []\nheights = []\nchans=[]\n\n# Iterate through the test_generator\nfor batch in test_generator:\n    image = batch[0]  # Assuming batch size is 1\n    width = image.shape[0]\n    height = image.shape[1]\n    chan = image.shape[2]\n    widths.append(width)\n    heights.append(height)\n    chans.append(chan)\n\n    # Optionally, you can break the loop if you've processed all images\n    if test_generator.batch_index == 0:\n        break\n\n# Create histograms for width and height\nplt.figure(figsize=(12, 6))\n\n# Plot the histogram for image widths\nplt.subplot(1, 3, 1)\nplt.hist(widths, bins=50, color='b', alpha=0.7)\nplt.xlabel('Width')\nplt.ylabel('Frequency')\nplt.title('Distribution of Image Widths')\n\n# Plot the histogram for image heights\nplt.subplot(1, 3, 2)\nplt.hist(heights, bins=50, color='g', alpha=0.7)\nplt.xlabel('Height')\nplt.ylabel('Frequency')\nplt.title('Distribution of Image Heights')\n\nplt.subplot(1, 3, 3)\nplt.hist(chans, bins=50, color='g', alpha=0.7)\nplt.xlabel('chans')\nplt.ylabel('Frequency')\nplt.title('Distribution of Image chans')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-07T02:34:59.818806Z","iopub.execute_input":"2023-11-07T02:34:59.819096Z","iopub.status.idle":"2023-11-07T02:36:38.464773Z","shell.execute_reply.started":"2023-11-07T02:34:59.819072Z","shell.execute_reply":"2023-11-07T02:36:38.463917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.applications import ResNet50\nfrom keras.layers import GlobalAveragePooling2D, Dense, Dropout\nfrom keras.models import Model\n\ndef create_model(input_shape, n_out):\n    input_tensor = Input(shape=input_shape)\n    base_model = ResNet50(weights=None, include_top=False, input_tensor=input_tensor)\n    \n    x = GlobalAveragePooling2D()(base_model.output)\n    x = Dropout(0.5)(x)\n    x = Dense(2048, activation='relu')(x)\n    x = Dropout(0.5)(x)\n    final_output = Dense(n_out, activation='softmax', name='final_output')(x)\n    model = Model(inputs=input_tensor, outputs=final_output)\n    \n    return model\n\nmodel = create_model(input_shape=(HEIGHT, WIDTH, CANAL), n_out=N_CLASSES)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-07T02:36:38.466145Z","iopub.execute_input":"2023-11-07T02:36:38.466499Z","iopub.status.idle":"2023-11-07T02:36:45.179469Z","shell.execute_reply.started":"2023-11-07T02:36:38.466465Z","shell.execute_reply":"2023-11-07T02:36:45.178667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in model.layers:\n    layer.trainable = False\n\nfor i in range(-5, 0):\n    model.layers[i].trainable = True\n\nmetric_list = [\"accuracy\"]\noptimizer = optimizers.Adam(lr=WARMUP_LEARNING_RATE)\nmodel.compile(optimizer=optimizer, loss=\"categorical_crossentropy\",  metrics=metric_list)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-11-07T02:36:45.180569Z","iopub.execute_input":"2023-11-07T02:36:45.180876Z","iopub.status.idle":"2023-11-07T02:36:45.632565Z","shell.execute_reply.started":"2023-11-07T02:36:45.180831Z","shell.execute_reply":"2023-11-07T02:36:45.631622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEP_SIZE_TRAIN = train_generator.n//train_generator.batch_size\nSTEP_SIZE_VALID = valid_generator.n//valid_generator.batch_size\n\nhistory_warmup = model.fit_generator(generator=train_generator,\n                              steps_per_epoch=STEP_SIZE_TRAIN,\n                              validation_data=valid_generator,\n                              validation_steps=STEP_SIZE_VALID,\n                              epochs=WARMUP_EPOCHS,\n                              verbose=1).history","metadata":{"execution":{"iopub.status.busy":"2023-11-07T02:36:45.63398Z","iopub.execute_input":"2023-11-07T02:36:45.634257Z","iopub.status.idle":"2023-11-07T03:14:47.397446Z","shell.execute_reply.started":"2023-11-07T02:36:45.634232Z","shell.execute_reply":"2023-11-07T03:14:47.396619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in model.layers:\n    layer.trainable = True\n\nes = EarlyStopping(monitor='val_loss', mode='min', patience=ES_PATIENCE, restore_best_weights=True, verbose=1)\nrlrop = ReduceLROnPlateau(monitor='val_loss', mode='min', patience=RLROP_PATIENCE, factor=DECAY_DROP, min_lr=1e-6, verbose=1)\n\ncallback_list = [es, rlrop]\noptimizer = optimizers.Adam(lr=LEARNING_RATE)\nmodel.compile(optimizer=optimizer, loss=\"binary_crossentropy\",  metrics=metric_list)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-11-07T03:14:47.398691Z","iopub.execute_input":"2023-11-07T03:14:47.399002Z","iopub.status.idle":"2023-11-07T03:14:47.868137Z","shell.execute_reply.started":"2023-11-07T03:14:47.398977Z","shell.execute_reply":"2023-11-07T03:14:47.867202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_finetunning = model.fit_generator(generator=train_generator,\n                                          steps_per_epoch=STEP_SIZE_TRAIN,\n                                          validation_data=valid_generator,\n                                          validation_steps=STEP_SIZE_VALID,\n                                          epochs=EPOCHS,\n                                          callbacks=callback_list,\n                                          verbose=1).history","metadata":{"execution":{"iopub.status.busy":"2023-11-07T03:14:47.869484Z","iopub.execute_input":"2023-11-07T03:14:47.86987Z","iopub.status.idle":"2023-11-07T09:32:28.967946Z","shell.execute_reply.started":"2023-11-07T03:14:47.869814Z","shell.execute_reply":"2023-11-07T09:32:28.967075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = {'loss': history_warmup['loss'] + history_finetunning['loss'], \n           'val_loss': history_warmup['val_loss'] + history_finetunning['val_loss'], \n           'accuracy': history_warmup['accuracy'] + history_finetunning['accuracy'], \n           'val_accuracy': history_warmup['val_accuracy'] + history_finetunning['val_accuracy']}\n\nsns.set_style(\"whitegrid\")\nfig, (ax1, ax2) = plt.subplots(2, 1, sharex='col', figsize=(20, 14))\n\nax1.plot(history['loss'], label='Train loss')\nax1.plot(history['val_loss'], label='Validation loss')\nax1.legend(loc='best')\nax1.set_title('Loss')\n\nax2.plot(history['accuracy'], label='Train Accuracy')\nax2.plot(history['val_accuracy'], label='Validation accuracy')\nax2.legend(loc='best')\nax2.set_title('Accuracy')\n\nplt.xlabel('Epochs')\nsns.despine()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-07T09:32:28.96925Z","iopub.execute_input":"2023-11-07T09:32:28.969549Z","iopub.status.idle":"2023-11-07T09:32:29.710857Z","shell.execute_reply.started":"2023-11-07T09:32:28.969525Z","shell.execute_reply":"2023-11-07T09:32:29.709897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"complete_datagen = ImageDataGenerator(rescale=1./255)\ncomplete_generator = complete_datagen.flow_from_dataframe(  \n        dataframe=train_new,\n        directory = \"../input/aptos2019-blindness-detection/train_images/\",\n        x_col=\"id_code\",\n        target_size=(HEIGHT, WIDTH),\n        batch_size=1,\n        shuffle=False,\n        class_mode=None)\n\nSTEP_SIZE_COMPLETE = complete_generator.n//complete_generator.batch_size\ntrain_preds = model.predict_generator(complete_generator, steps=STEP_SIZE_COMPLETE)\ntrain_preds = [np.argmax(pred) for pred in train_preds]","metadata":{"execution":{"iopub.status.busy":"2023-11-07T09:32:29.712068Z","iopub.execute_input":"2023-11-07T09:32:29.712356Z","iopub.status.idle":"2023-11-07T09:51:18.492025Z","shell.execute_reply.started":"2023-11-07T09:32:29.712332Z","shell.execute_reply":"2023-11-07T09:51:18.490882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = ['0 - No DR', '1 - Mild', '2 - Moderate', '3 - Severe', '4 - Proliferative DR']\ncnf_matrix = confusion_matrix(train_new['diagnosis'].astype('int'), train_preds)\ncnf_matrix_norm = cnf_matrix.astype('float') / cnf_matrix.sum(axis=1)[:, np.newaxis]\ndf_cm = pd.DataFrame(cnf_matrix_norm, index=labels, columns=labels)\nplt.figure(figsize=(5, 2))\nsns.heatmap(df_cm, annot=True, fmt='.2f', cmap=\"Blues\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-07T09:51:18.493329Z","iopub.execute_input":"2023-11-07T09:51:18.493633Z","iopub.status.idle":"2023-11-07T09:51:18.890097Z","shell.execute_reply.started":"2023-11-07T09:51:18.493607Z","shell.execute_reply":"2023-11-07T09:51:18.889199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\n\ncr = classification_report(train_preds, train_new['diagnosis'].astype('int'))\nprint(cr)","metadata":{"execution":{"iopub.status.busy":"2023-11-07T09:51:18.891189Z","iopub.execute_input":"2023-11-07T09:51:18.892301Z","iopub.status.idle":"2023-11-07T09:51:18.924187Z","shell.execute_reply.started":"2023-11-07T09:51:18.892269Z","shell.execute_reply":"2023-11-07T09:51:18.923176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Train Cohen Kappa score: %.3f\" % cohen_kappa_score(train_preds, train_new['diagnosis'].astype('int'), weights='quadratic'))","metadata":{"execution":{"iopub.status.busy":"2023-11-07T09:51:18.925623Z","iopub.execute_input":"2023-11-07T09:51:18.926651Z","iopub.status.idle":"2023-11-07T09:51:18.938623Z","shell.execute_reply.started":"2023-11-07T09:51:18.926615Z","shell.execute_reply":"2023-11-07T09:51:18.937699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_generator.reset()\nSTEP_SIZE_TEST = test_generator.n//test_generator.batch_size\npreds = model.predict_generator(test_generator, steps=STEP_SIZE_TEST)\npredictions = [np.argmax(pred) for pred in preds]","metadata":{"execution":{"iopub.status.busy":"2023-11-07T09:51:18.939785Z","iopub.execute_input":"2023-11-07T09:51:18.940738Z","iopub.status.idle":"2023-11-07T09:52:35.822893Z","shell.execute_reply.started":"2023-11-07T09:51:18.940689Z","shell.execute_reply":"2023-11-07T09:52:35.822052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filenames = test_generator.filenames\nresults = pd.DataFrame({'id_code':filenames, 'diagnosis':predictions})\nresults['id_code'] = results['id_code'].map(lambda x: str(x)[:-4])\nresults.to_csv('submission.csv',index=False)\nresults.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-11-07T09:52:35.824032Z","iopub.execute_input":"2023-11-07T09:52:35.824318Z","iopub.status.idle":"2023-11-07T09:52:35.852857Z","shell.execute_reply.started":"2023-11-07T09:52:35.824294Z","shell.execute_reply":"2023-11-07T09:52:35.851981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, ax = plt.subplots(figsize=(5, 4.2))\nax = sns.countplot(x=\"diagnosis\", data=results, palette=\"GnBu_d\")\nsns.despine()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-07T09:52:35.853805Z","iopub.execute_input":"2023-11-07T09:52:35.854088Z","iopub.status.idle":"2023-11-07T09:52:36.103287Z","shell.execute_reply.started":"2023-11-07T09:52:35.854065Z","shell.execute_reply":"2023-11-07T09:52:36.102244Z"},"trusted":true},"execution_count":null,"outputs":[]}]}