{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport kagglehub","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-10-04T07:32:09.870269Z","iopub.execute_input":"2026-10-04T07:32:09.870719Z","iopub.status.idle":"2026-10-04T07:32:09.875388Z","shell.execute_reply.started":"2026-10-04T07:32:09.870688Z","shell.execute_reply":"2026-10-04T07:32:09.874356Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nimport pandas as pd\nfrom IPython.display import display, Image\n\n# Kaggle datasets are mounted under /kaggle/input\nINPUT_DIR = Path(\"/kaggle/input\")\n\n# Find the labels file; APTOS 2019 is commonly named train.csv\ncsv_files = list(INPUT_DIR.rglob(\"train.csv\"))\nif not csv_files:\n    raise FileNotFoundError(\n        \"Couldn't find train.csv under /kaggle/input. \"\n        \"Use Kaggle's Add Input button to attach the APTOS dataset.\"\n    )\n\nlabels_path = csv_files[0]\nlabels = pd.read_csv(labels_path)\n\nprint(f\"Labels file: {labels_path}\")\nprint(f\"Rows: {len(labels):,}\")\nprint(\"Columns:\", labels.columns.tolist())\ndisplay(labels.head())\n\n# Locate the folder containing the training images\nimage_extensions = {\".png\", \".jpg\", \".jpeg\"}\nimage_files = [\n    p for p in INPUT_DIR.rglob(\"*\")\n    if p.is_file() and p.suffix.lower() in image_extensions\n]\n\n# APTOS image names are usually stored in the id_code column\nif \"id_code\" in labels.columns:\n    known_ids = set(labels[\"id_code\"].astype(str))\n    matched_images = [p for p in image_files if p.stem in known_ids]\nelse:\n    matched_images = []\n\nif not matched_images:\n    print(\"\\nCouldn't automatically match image files to labels.\")\n    print(\"Image files found:\", len(image_files))\n    print(\"First few image paths:\")\n    for path in image_files[:10]:\n        print(path)\nelse:\n    image_by_id = {p.stem: p for p in matched_images}\n    labels[\"image_path\"] = labels[\"id_code\"].astype(str).map(image_by_id)\n\n    print(f\"\\nMatched images: {labels['image_path'].notna().sum():,} / {len(labels):,}\")\n    display(labels[[\"id_code\", \"diagnosis\", \"image_path\"]].head())\n\n    # Show a few sample images with their labels\n    for _, row in labels.dropna(subset=[\"image_path\"]).head(3).iterrows():\n        print(f\"DR grade: {row['diagnosis']} — {row['id_code']}\")\n        display(Image(filename=str(row[\"image_path\"]), width=400))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-04T07:32:20.198251Z","iopub.execute_input":"2026-10-04T07:32:20.198669Z","iopub.status.idle":"2026-10-04T07:32:24.409463Z","shell.execute_reply.started":"2026-10-04T07:32:20.19864Z","shell.execute_reply":"2026-10-04T07:32:24.407745Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grade_counts = labels[\"diagnosis\"].value_counts().sort_index()\n\nprint(\"Images per DR grade:\")\ndisplay(grade_counts.rename_axis(\"Grade\").to_frame(\"Number of images\"))\n\ngrade_counts.plot(\n    kind=\"bar\",\n    title=\"APTOS images by DR grade\",\n    xlabel=\"DR grade\",\n    ylabel=\"Number of images\",\n    rot=0\n);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-04T07:32:34.175051Z","iopub.execute_input":"2026-10-04T07:32:34.17581Z","iopub.status.idle":"2026-10-04T07:32:34.343174Z","shell.execute_reply.started":"2026-10-04T07:32:34.175778Z","shell.execute_reply":"2026-10-04T07:32:34.34215Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\n\nIMAGE_SIZE = 224\nBATCH_SIZE = 16\nSEED = 42\n\ndef load_image(path, label):\n    image = tf.io.read_file(path)\n    image = tf.io.decode_image(\n        image,\n        channels=3,\n        expand_animations=False\n    )\n    image.set_shape([None, None, 3])\n    image = tf.image.resize(image, [IMAGE_SIZE, IMAGE_SIZE])\n    image = tf.cast(image, tf.float32)\n\n    return image, tf.cast(label, tf.int32)\n\n\ndef make_dataset(dataframe, training=False):\n    paths = dataframe[\"image_path\"].astype(str).to_numpy()\n    grades = dataframe[\"diagnosis\"].to_numpy(dtype=\"int32\")\n\n    dataset = tf.data.Dataset.from_tensor_slices((paths, grades))\n\n    if training:\n        dataset = dataset.shuffle(\n            buffer_size=len(dataframe),\n            seed=SEED,\n            reshuffle_each_iteration=True\n        )\n\n    dataset = dataset.map(\n        load_image,\n        num_parallel_calls=tf.data.AUTOTUNE\n    )\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(tf.data.AUTOTUNE)\n\n    return dataset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-04T07:35:46.476493Z","iopub.execute_input":"2026-10-04T07:35:46.476875Z","iopub.status.idle":"2026-10-04T07:35:46.484295Z","shell.execute_reply.started":"2026-10-04T07:35:46.476847Z","shell.execute_reply":"2026-10-04T07:35:46.483548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Previous approach did not give very fruitful sensitivity, specificity and overall accuracy, so now we try different split\n# 70%- train, 15%-validation, 15%-test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T07:53:44.68705Z","iopub.execute_input":"2026-10-03T07:53:44.687724Z","iopub.status.idle":"2026-10-03T07:53:44.692967Z","shell.execute_reply.started":"2026-10-03T07:53:44.68769Z","shell.execute_reply":"2026-10-03T07:53:44.692189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nSEED = 42\n\n# 70% train, 15% validation, 15% test\ntrain_df, temp_df = train_test_split(\n    labels,\n    test_size=0.30,\n    random_state=SEED,\n    stratify=labels[\"diagnosis\"]\n)\n\nval_df, test_df = train_test_split(\n    temp_df,\n    test_size=0.50,\n    random_state=SEED,\n    stratify=temp_df[\"diagnosis\"]\n)\n\nprint(\"Rows:\", len(train_df), len(val_df), len(test_df))\nprint(\"\\nGrade counts in each split:\")\nfor name, df in [(\"Train\", train_df), (\"Validation\", val_df), (\"Test\", test_df)]:\n    print(f\"\\n{name}\")\n    display(df[\"diagnosis\"].value_counts().sort_index())\n\n# Rebuild datasets from these new splits\ntrain_ds = make_dataset(train_df, training=True)\nval_ds = make_dataset(val_df)\ntest_ds = make_dataset(test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-04T07:35:51.982625Z","iopub.execute_input":"2026-10-04T07:35:51.983651Z","iopub.status.idle":"2026-10-04T07:35:52.09782Z","shell.execute_reply.started":"2026-10-04T07:35:51.983609Z","shell.execute_reply":"2026-10-04T07:35:52.096825Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.utils.class_weight import compute_class_weight\n\nclass_ids = np.array([0, 1, 2, 3, 4])\n\nweights = compute_class_weight(\n    class_weight=\"balanced\",\n    classes=class_ids,\n    y=train_df[\"diagnosis\"].to_numpy()\n)\n\nclass_weights = {int(c): float(w) for c, w in zip(class_ids, weights)}\nprint(class_weights)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-04T07:36:22.680106Z","iopub.execute_input":"2026-10-04T07:36:22.681125Z","iopub.status.idle":"2026-10-04T07:36:22.689037Z","shell.execute_reply.started":"2026-10-04T07:36:22.681078Z","shell.execute_reply":"2026-10-04T07:36:22.688264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\n\nIMAGE_SIZE = 224\n\nbase_weighted = tf.keras.applications.EfficientNetB0(\n    include_top=False,\n    weights=\"imagenet\",\n    input_shape=(IMAGE_SIZE, IMAGE_SIZE, 3)\n)\nbase_weighted.trainable = False\n\ninputs = tf.keras.Input(shape=(IMAGE_SIZE, IMAGE_SIZE, 3))\nx = base_weighted(inputs, training=False)\nx = tf.keras.layers.GlobalAveragePooling2D()(x)\nx = tf.keras.layers.Dropout(0.3)(x)\noutputs = tf.keras.layers.Dense(5, activation=\"softmax\")(x)\n\nmodel_weighted = tf.keras.Model(inputs, outputs)\n\nmodel_weighted.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=1e-3),\n    loss=\"sparse_categorical_crossentropy\",\n    metrics=[\"accuracy\"]\n)\n\nmodel_weighted.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-04T07:45:47.258029Z","iopub.execute_input":"2026-10-04T07:45:47.258377Z","iopub.status.idle":"2026-10-04T07:45:48.39719Z","shell.execute_reply.started":"2026-10-04T07:45:47.25835Z","shell.execute_reply":"2026-10-04T07:45:48.396548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"checkpoint_cb = tf.keras.callbacks.ModelCheckpoint(\n    \"/kaggle/working/aptos_weighted_latest.keras\",\n    save_freq=\"epoch\",\n    save_best_only=False,\n    save_weights_only=False\n)\n\n# Starts at Epoch 1 and saves a full checkpoint after each completed epoch\nhistory_more = model_weighted.fit(\n    train_ds,\n    validation_data=val_ds,\n    class_weight=class_weights,\n    epochs=10,\n    callbacks=[checkpoint_cb]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-04T07:46:11.580343Z","iopub.execute_input":"2026-10-04T07:46:11.580801Z","iopub.status.idle":"2026-10-04T07:50:41.20529Z","shell.execute_reply.started":"2026-10-04T07:46:11.580772Z","shell.execute_reply":"2026-10-04T07:50:41.20179Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.metrics import classification_report, confusion_matrix\n\ny_val = val_df[\"diagnosis\"].to_numpy(dtype=np.int64)\nval_probabilities_weighted = model_weighted.predict(val_ds)\ny_pred_weighted = np.argmax(val_probabilities_weighted, axis=1)\n\nprint(classification_report(\n    y_val,\n    y_pred_weighted,\n    labels=[0, 1, 2, 3, 4],\n    target_names=[\"Grade 0\", \"Grade 1\", \"Grade 2\", \"Grade 3\", \"Grade 4\"],\n    zero_division=0\n))\n\nprint(\"Confusion matrix:\")\nprint(confusion_matrix(y_val, y_pred_weighted, labels=[0, 1, 2, 3, 4]))\n\nactual_ref = y_val >= 2\npredicted_ref = y_pred_weighted >= 2\n\ntp = np.sum(actual_ref & predicted_ref)\nfn = np.sum(actual_ref & ~predicted_ref)\ntn = np.sum(~actual_ref & ~predicted_ref)\nfp = np.sum(~actual_ref & predicted_ref)\n\nprint(f\"Referable DR sensitivity: {tp / (tp + fn):.3f}\")\nprint(f\"Referable DR specificity: {tn / (tn + fp):.3f}\")\nprint(f\"TP={tp}, FN={fn}, TN={tn}, FP={fp}\")\n\nmodel_weighted.save(\n    \"/kaggle/working/aptos_efficientnetb0_weighted_final.keras\"\n)\nprint(\"Weighted model saved.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-04T07:50:49.858468Z","iopub.execute_input":"2026-10-04T07:50:49.858735Z","iopub.status.idle":"2026-10-04T07:51:19.431452Z","shell.execute_reply.started":"2026-10-04T07:50:49.858711Z","shell.execute_reply":"2026-10-04T07:51:19.430678Z"}},"outputs":[],"execution_count":null}]}