{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import Packages\n\nThis section imports necessary Python packages for data manipulation, machine learning, and image processing.","metadata":{}},{"cell_type":"code","source":"import os\nimport re\nimport sys\nfrom tqdm import tqdm\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport seaborn as sns\nfrom math import ceil\nfrom PIL import Image\nfrom matplotlib import pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom keras.applications.inception_v3 import InceptionV3, preprocess_input\nfrom keras.applications.xception import Xception\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Sequential, Model\nfrom keras.layers import Input, Dense, GlobalAveragePooling2D, Dropout\nfrom keras.optimizers import RMSprop, Adam, SGD\nfrom keras.callbacks import ModelCheckpoint, ReduceLROnPlateau, EarlyStopping\nfrom IPython.display import FileLink","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2023-07-31T05:04:25.553187Z","iopub.execute_input":"2023-07-31T05:04:25.554112Z","iopub.status.idle":"2023-07-31T05:04:25.562495Z","shell.execute_reply.started":"2023-07-31T05:04:25.554066Z","shell.execute_reply":"2023-07-31T05:04:25.561305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Check Dataset\n\nThis section checks the dataset paths and loads the train and test CSV files into pandas DataFrames. It also prints the number of train and test images. Additionally, it plots a histogram to show the number of images per diagnosis category in the training data.","metadata":{}},{"cell_type":"code","source":"DATA_PATH = '../input/aptos2019-blindness-detection'\n\nTRAIN_IMG_PATH = os.path.join(DATA_PATH, 'train_images')\nTEST_IMG_PATH = os.path.join(DATA_PATH, 'test_images')\nTRAIN_LABEL_PATH = os.path.join(DATA_PATH, 'train.csv')\nTEST_LABEL_PATH = os.path.join(DATA_PATH, 'test.csv')\n\ndf_train = pd.read_csv(TRAIN_LABEL_PATH)\ndf_test = pd.read_csv(TEST_LABEL_PATH)\n\nprint('num of train images ', len(os.listdir(TRAIN_IMG_PATH)))\nprint('num of test images  ', len(os.listdir(TEST_IMG_PATH)))","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.status.busy":"2023-07-31T05:04:25.564831Z","iopub.execute_input":"2023-07-31T05:04:25.565456Z","iopub.status.idle":"2023-07-31T05:04:25.591600Z","shell.execute_reply.started":"2023-07-31T05:04:25.565395Z","shell.execute_reply":"2023-07-31T05:04:25.590694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the counts of each diagnosis category\ndiagnosis_counts = df_train['diagnosis'].value_counts()\n\n# Set the figure size\nplt.figure(figsize=(10, 5))\n\n# Create a bar plot\nplt.bar(diagnosis_counts.index, diagnosis_counts.values, align='center', width=0.8)\n\n# Set labels and title\nplt.xlabel('Diagnosis')\nplt.ylabel('Count')\nplt.title('Number of data per each diagnosis')\n\n# Set the x-axis ticks\nplt.xticks(range(len(diagnosis_counts.index)), diagnosis_counts.index)\n\n# Show the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-31T10:59:42.533668Z","iopub.execute_input":"2023-07-31T10:59:42.534078Z","iopub.status.idle":"2023-07-31T10:59:42.802273Z","shell.execute_reply.started":"2023-07-31T10:59:42.534046Z","shell.execute_reply":"2023-07-31T10:59:42.801274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Split Training Set\n\nThis section splits the training data into a train set and a validation set. The validation set is used for evaluating the model during the training process.","metadata":{}},{"cell_type":"code","source":"# Convert 'diagnosis' column to string type\ndf_train['diagnosis'] = df_train['diagnosis'].astype(str)\n\n# Append '.png' extension to 'id_code' if not already present\ndf_train['id_code'] = df_train['id_code'].apply(lambda x: x + '.png' if x.split('.')[-1] != 'png' else x)\ndf_test['id_code'] = df_test['id_code'].apply(lambda x: x + '.png' if x.split('.')[-1] != 'png' else x)\n\n# Split the training data into training and validation sets\ntrain_data = np.arange(df_train.shape[0])\ntrain_idx, val_idx = train_test_split(train_data, train_size=0.8, random_state=2019)\n\nX_train = df_train.iloc[train_idx, :]\nX_val = df_train.iloc[val_idx, :]\nX_test = df_test\n\nprint(X_train.shape)\nprint(X_val.shape)\nprint(X_test.shape)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T05:04:25.866297Z","iopub.execute_input":"2023-07-31T05:04:25.866900Z","iopub.status.idle":"2023-07-31T05:04:25.885922Z","shell.execute_reply.started":"2023-07-31T05:04:25.866866Z","shell.execute_reply":"2023-07-31T05:04:25.884987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pre-trained Model Preparation\n\nThis section defines a function to create a machine learning model. The model architecture is based on InceptionV3, a pre-trained model on ImageNet. The last layers of the model are customly added for this specific problem.","metadata":{}},{"cell_type":"code","source":"num_classes = 5\nimg_size = (299, 299, 3)\nnb_train_samples = len(X_train)\nnb_validation_samples = len(X_val)\nnb_test_samples = len(X_test)\nepochs = 50\nbatch_size = 32\n\ntrain_datagen = ImageDataGenerator(\n    horizontal_flip=True,\n    vertical_flip=True,\n    width_shift_range=0.1,\n    height_shift_range=0.1,\n    brightness_range=[0.5, 1.5],\n    rescale=1./255\n)\nval_datagen = ImageDataGenerator(\n    rescale=1./255\n)\n# Apply TTA\ntest_datagen = ImageDataGenerator(\n    horizontal_flip=True,\n    vertical_flip=True,\n    width_shift_range=0.1,\n    height_shift_range=0.1,\n    brightness_range=[0.5, 1.5],\n    rescale=1./255\n)\n\ntrain_generator = train_datagen.flow_from_dataframe(\n    dataframe=X_train, \n    directory=TRAIN_IMG_PATH,\n    x_col='id_code',\n    y_col='diagnosis',\n    target_size=img_size[:2],\n    color_mode='rgb',\n    class_mode='categorical',\n    batch_size=batch_size,\n    seed=2019\n)\nvalidation_generator = val_datagen.flow_from_dataframe(\n    dataframe=X_val, \n    directory=TRAIN_IMG_PATH,\n    x_col='id_code',\n    y_col='diagnosis',\n    target_size=img_size[:2],\n    color_mode='rgb',\n    class_mode='categorical',\n    batch_size=batch_size,\n    shuffle=False,\n    seed=2019\n)\ntest_generator = test_datagen.flow_from_dataframe(\n    dataframe=X_test,\n    directory=TEST_IMG_PATH,\n    x_col='id_code',\n    y_col=None,\n    target_size= img_size[:2],\n    color_mode='rgb',\n    class_mode=None,\n    batch_size=batch_size,\n    shuffle=False,\n    seed=2019\n)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T05:04:25.888906Z","iopub.execute_input":"2023-07-31T05:04:25.889635Z","iopub.status.idle":"2023-07-31T05:04:28.443276Z","shell.execute_reply.started":"2023-07-31T05:04:25.889603Z","shell.execute_reply":"2023-07-31T05:04:28.442387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training Process\n\nThis section sets up the training process, including defining the checkpoint and early stopping callbacks. Then, it starts training the model using the fit_generator function.","metadata":{}},{"cell_type":"code","source":"def get_model(model_type, input_shape, num_classes):\n    input_tensor = Input(shape=input_shape)\n    if model_type == 'InceptionV3':\n        base_model = InceptionV3(include_top=False, weights='imagenet', input_tensor=input_tensor)\n    elif model_type == 'Xception':\n        base_model = Xception(include_top=False, weights='imagenet', input_tensor=input_tensor)\n    else:\n        print(\"Invalid model_type; defaulting to InceptionV3\")\n        base_model = InceptionV3(include_top=False, weights=None, input_tensor=input_tensor)\n        base_model.load_weights(filepath=file_path)\n\n    x = base_model.output\n    x = GlobalAveragePooling2D()(x)\n    x = Dense(1024, activation='relu')(x)\n    x = Dropout(0.25)(x)\n    output_tensor = Dense(num_classes, activation='softmax')(x)\n    \n    model = Model(inputs=input_tensor, outputs=output_tensor)\n    \n    optimizer = RMSprop(learning_rate=1e-4)\n    model.compile(loss='categorical_crossentropy', optimizer=optimizer, metrics=['accuracy'])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-07-31T06:30:22.533436Z","iopub.execute_input":"2023-07-31T06:30:22.533834Z","iopub.status.idle":"2023-07-31T06:30:22.542796Z","shell.execute_reply.started":"2023-07-31T06:30:22.533804Z","shell.execute_reply":"2023-07-31T06:30:22.541629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training Inception and Xception Models\n\nThis section trains the Inception and Xception models using early stopping and learning rate reduction on plateau callbacks, saving the best models based on validation loss.","metadata":{}},{"cell_type":"code","source":"LOG_DIR = './logs'\nif not os.path.isdir(LOG_DIR):\n    os.mkdir(LOG_DIR)\nelse:\n    pass","metadata":{"execution":{"iopub.status.busy":"2023-07-31T06:30:26.570826Z","iopub.execute_input":"2023-07-31T06:30:26.571874Z","iopub.status.idle":"2023-07-31T06:30:26.577871Z","shell.execute_reply.started":"2023-07-31T06:30:26.571828Z","shell.execute_reply":"2023-07-31T06:30:26.576575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the input shape and number of classes for your specific problem\ninput_shape = (299, 299, 3)  # for example, if you're using 299x299 RGB images\nnum_classes = 5  # replace with the actual number of classes\n\n# Instantiate the models\nmodel_inception = get_model('InceptionV3', input_shape, num_classes)\nmodel_xception = get_model('Xception', input_shape, num_classes)\n\n# Inception Model\nCKPT_PATH_INCEPTION = LOG_DIR + '/inception_best_model.hdf5'\ncheckPoint_inception = ModelCheckpoint(filepath=CKPT_PATH_INCEPTION, monitor='val_loss', verbose=1, save_best_only=True, mode='min')\nreduceLROnPlateau_inception = ReduceLROnPlateau(monitor='val_loss', factor=0.1, patience=2, min_lr=0.0000001, verbose=1, mode='min')\nearlyStopping_inception = EarlyStopping(monitor='val_loss', patience=5, verbose=1, mode='min')\nhistory_inception = model_inception.fit(train_generator, steps_per_epoch=ceil(nb_train_samples/batch_size), epochs=epochs, validation_data=validation_generator, validation_steps=ceil(nb_validation_samples/batch_size), callbacks=[checkPoint_inception, reduceLROnPlateau_inception, earlyStopping_inception], verbose=1)\n\n# Xception Model\nCKPT_PATH_XCEPTION = LOG_DIR + '/xception_best_model.hdf5'\ncheckPoint_xception = ModelCheckpoint(filepath=CKPT_PATH_XCEPTION, monitor='val_loss', verbose=1, save_best_only=True, mode='min')\nreduceLROnPlateau_xception = ReduceLROnPlateau(monitor='val_loss', factor=0.1, patience=2, min_lr=0.0000001, verbose=1, mode='min')\nearlyStopping_xception = EarlyStopping(monitor='val_loss', patience=5, verbose=1, mode='min')\nhistory_xception = model_xception.fit(train_generator, steps_per_epoch=ceil(nb_train_samples/batch_size), epochs=epochs, validation_data=validation_generator, validation_steps=ceil(nb_validation_samples/batch_size), callbacks=[checkPoint_xception, reduceLROnPlateau_xception, earlyStopping_xception], verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T06:31:12.053442Z","iopub.execute_input":"2023-07-31T06:31:12.053854Z","iopub.status.idle":"2023-07-31T10:12:51.391954Z","shell.execute_reply.started":"2023-07-31T06:31:12.053823Z","shell.execute_reply":"2023-07-31T10:12:51.390615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission\n\nThis section prepares a submission file for the competition based on the predictions. It also plots a histogram of the predicted labels and prints out the counts of each diagnosis in the predictions.","metadata":{}},{"cell_type":"code","source":"# For Inception model\nacc_inception = history_inception.history['accuracy']\nval_acc_inception = history_inception.history['val_accuracy']\n\nplt.figure(figsize=(10, 5))\nplt.subplot(1, 2, 1)\nplt.plot(acc_inception)\nplt.plot(val_acc_inception)\nplt.title('Inception Model')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.legend(['Train', 'Val'], loc='upper left')\n\n# For Xception model\nacc_xception = history_xception.history['accuracy']\nval_acc_xception = history_xception.history['val_accuracy']\n\nplt.subplot(1, 2, 2)\nplt.plot(acc_xception)\nplt.plot(val_acc_xception)\nplt.title('Xception Model')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.legend(['Train', 'Val'], loc='upper left')\n\nplt.tight_layout()\nplt.show()\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-07-31T10:26:46.819289Z","iopub.execute_input":"2023-07-31T10:26:46.819689Z","iopub.status.idle":"2023-07-31T10:26:47.394217Z","shell.execute_reply.started":"2023-07-31T10:26:46.819658Z","shell.execute_reply":"2023-07-31T10:26:47.393266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Inception Model\nloss_inception = history_inception.history['loss']\nval_loss_inception = history_inception.history['val_loss']\n\n# Xception Model\nloss_xception = history_xception.history['loss']\nval_loss_xception = history_xception.history['val_loss']\n\nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(10, 5))\n\n# Plot for Inception Model\nplt.subplot(1, 2, 1)\nplt.plot(loss_inception)\nplt.plot(val_loss_inception)\nplt.title('Inception Model')\nplt.ylabel('Loss')\nplt.xlabel('Epoch')\nplt.legend(['Train', 'Val'], loc='upper left')\n\n# Plot for Xception Model\nplt.subplot(1, 2, 2)\nplt.plot(loss_xception)\nplt.plot(val_loss_xception)\nplt.title('Xception Model')\nplt.ylabel('Loss')\nplt.xlabel('Epoch')\nplt.legend(['Train', 'Val'], loc='upper left')\n\nplt.tight_layout()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-07-31T10:26:50.214951Z","iopub.execute_input":"2023-07-31T10:26:50.215307Z","iopub.status.idle":"2023-07-31T10:26:50.800160Z","shell.execute_reply.started":"2023-07-31T10:26:50.215277Z","shell.execute_reply":"2023-07-31T10:26:50.799225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_best_model(model_name):\n    # Get all saved model weights for the specific model\n    all_models = [f for f in os.listdir(LOG_DIR) if re.search(model_name, f)]\n    \n    # Extract the validation loss from the filenames\n    losses = [float(re.search('(\\d+\\.\\d+).hdf5', f).group(1)) for f in all_models]\n    \n    # Get the filename of the model with the lowest validation loss\n    best_model = all_models[losses.index(min(losses))]\n    \n    return os.path.join(LOG_DIR, best_model)\n\nbest_model_inception = os.path.join(LOG_DIR, 'inception_best_model.hdf5')\nbest_model_xception = os.path.join(LOG_DIR, 'xception_best_model.hdf5')\n\n# Load the best weights into the models\nmodel_inception.load_weights(best_model_inception)\nmodel_xception.load_weights(best_model_xception)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T10:29:40.786521Z","iopub.execute_input":"2023-07-31T10:29:40.787481Z","iopub.status.idle":"2023-07-31T10:29:42.072101Z","shell.execute_reply.started":"2023-07-31T10:29:40.787442Z","shell.execute_reply":"2023-07-31T10:29:42.071079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test-Time Augmentation (TTA)","metadata":{}},{"cell_type":"code","source":"# Apply TTA\npreds_tta_inception = []\npreds_tta_xception = []\ntta_steps = 5\nfor i in tqdm(range(tta_steps)):\n    test_generator.reset()\n    preds_inception = model_inception.predict(test_generator, steps=ceil(nb_test_samples/batch_size))\n    preds_xception = model_xception.predict(test_generator, steps=ceil(nb_test_samples/batch_size))\n    preds_tta_inception.append(preds_inception)\n    preds_tta_xception.append(preds_xception)\n\npreds_mean_inception = np.mean(preds_tta_inception, axis=0)\npreds_mean_xception = np.mean(preds_tta_xception, axis=0)\npreds_mean = (preds_mean_inception + preds_mean_xception) / 2.0\n\npredicted_class_indices = np.argmax(preds_mean, axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T10:29:58.993824Z","iopub.execute_input":"2023-07-31T10:29:58.994220Z","iopub.status.idle":"2023-07-31T10:49:22.368727Z","shell.execute_reply.started":"2023-07-31T10:29:58.994191Z","shell.execute_reply":"2023-07-31T10:49:22.367561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv(os.path.join(DATA_PATH, 'sample_submission.csv'))\nsubmission['diagnosis'] = predicted_class_indices\nsubmission.to_csv(\"submission.csv\", index=False)\nsubmission.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T10:53:07.018450Z","iopub.execute_input":"2023-07-31T10:53:07.018826Z","iopub.status.idle":"2023-07-31T10:53:07.041802Z","shell.execute_reply.started":"2023-07-31T10:53:07.018797Z","shell.execute_reply":"2023-07-31T10:53:07.040512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Analysis of Diagnosis Distribution in the Dataset\n\nHere we generates a histogram to visualize the distribution of different diagnosis labels in the dataset. The submission DataFrame contains a column called 'diagnosis', representing the diagnosis labels for each data entry. The histogram displays the count of data points for each diagnosis, and the X-axis represents the diagnosis labels from 0 to 4.\n","metadata":{}},{"cell_type":"code","source":"plt.hist(submission['diagnosis'], bins=range(6), align='left', rwidth=0.8)\nplt.xlabel('Diagnosis')\nplt.ylabel('Count')\nplt.title('Number of data per each diagnosis')\nplt.xticks(range(5))\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-07-31T11:02:46.301653Z","iopub.execute_input":"2023-07-31T11:02:46.302047Z","iopub.status.idle":"2023-07-31T11:02:46.562267Z","shell.execute_reply.started":"2023-07-31T11:02:46.302016Z","shell.execute_reply":"2023-07-31T11:02:46.561301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Counting Diagnoses Manually\nFollowing the visualization, the code further computes the count of data points for each diagnosis class individually. This manual count is useful for gaining insight into the dataset's class distribution.","metadata":{}},{"cell_type":"code","source":"diagnosis_0 = 0\ndiagnosis_1 = 1\ndiagnosis_2 = 2\ndiagnosis_3 = 3\ndiagnosis_4 = 4\nfor idx in range(len(submission['diagnosis'])):\n    if submission['diagnosis'][idx] == 0:\n        diagnosis_0 += 1\n    elif submission['diagnosis'][idx] == 1:\n        diagnosis_1 += 1\n    elif submission['diagnosis'][idx] == 2:\n        diagnosis_2 += 1\n    elif submission['diagnosis'][idx] == 3:\n        diagnosis_3 += 1\n    elif submission['diagnosis'][idx] == 4:\n        diagnosis_4 += 1\nprint(\"  0 - No DR              {}\".format(diagnosis_0))\nprint(\"  1 - Mild               {}\".format(diagnosis_1))\nprint(\"  2 - Moderate           {}\".format(diagnosis_2))\nprint(\"  3 - Severe             {}\".format(diagnosis_3))\nprint(\"  4 - Proliferative DR   {}\".format(diagnosis_4))","metadata":{"execution":{"iopub.status.busy":"2023-07-31T11:02:48.710834Z","iopub.execute_input":"2023-07-31T11:02:48.711204Z","iopub.status.idle":"2023-07-31T11:02:48.776917Z","shell.execute_reply.started":"2023-07-31T11:02:48.711175Z","shell.execute_reply":"2023-07-31T11:02:48.775781Z"},"trusted":true},"execution_count":null,"outputs":[]}]}