{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":14774,"databundleVersionId":875431,"isSourceIdPinned":false}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nos.environ[\"TF_CPP_MIN_LOG_LEVEL\"] = \"3\"\nos.environ[\"TF_ENABLE_ONEDNN_OPTS\"] = \"0\"\nos.environ[\"TF_XLA_FLAGS\"] = \"--tf_xla_enable_xla_devices=false\"\n\nimport tensorflow as tf\nimport absl.logging\n\ntf.get_logger().setLevel('ERROR')\nabsl.logging.set_verbosity(absl.logging.ERROR)\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T16:26:04.910810Z","iopub.execute_input":"2026-03-31T16:26:04.911584Z","iopub.status.idle":"2026-03-31T16:26:32.859508Z","shell.execute_reply.started":"2026-03-31T16:26:04.911550Z","shell.execute_reply":"2026-03-31T16:26:32.858813Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1. IMPORTS","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import cohen_kappa_score\nfrom scipy.optimize import minimize","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T16:26:55.303166Z","iopub.execute_input":"2026-03-31T16:26:55.303710Z","iopub.status.idle":"2026-03-31T16:26:55.423845Z","shell.execute_reply.started":"2026-03-31T16:26:55.303672Z","shell.execute_reply":"2026-03-31T16:26:55.423027Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. CONFIG","metadata":{}},{"cell_type":"code","source":"IMG_SIZE = 224        \nBATCH_SIZE = 8\nEPOCHS = 15\n\nCSV_PATH = \"/kaggle/input/competitions/aptos2019-blindness-detection/train.csv\"\nIMAGE_DIR = \"/kaggle/input/competitions/aptos2019-blindness-detection/train_images\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T16:26:59.095733Z","iopub.execute_input":"2026-03-31T16:26:59.096484Z","iopub.status.idle":"2026-03-31T16:26:59.100127Z","shell.execute_reply.started":"2026-03-31T16:26:59.096452Z","shell.execute_reply":"2026-03-31T16:26:59.099383Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. LOAD DATA","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(CSV_PATH)\n\ndf['id_code'] = df['id_code'].astype(str) + \".png\"\ndf['diagnosis'] = df['diagnosis'].astype(float) / 4.0  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T16:27:25.061273Z","iopub.execute_input":"2026-03-31T16:27:25.062002Z","iopub.status.idle":"2026-03-31T16:27:25.084855Z","shell.execute_reply.started":"2026-03-31T16:27:25.061973Z","shell.execute_reply":"2026-03-31T16:27:25.084288Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. SPLIT","metadata":{}},{"cell_type":"code","source":"train_df, val_df = train_test_split(\n    df,\n    test_size=0.2,\n    stratify=df['diagnosis'],\n    random_state=42\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T16:27:28.592613Z","iopub.execute_input":"2026-03-31T16:27:28.592917Z","iopub.status.idle":"2026-03-31T16:27:28.609632Z","shell.execute_reply.started":"2026-03-31T16:27:28.592892Z","shell.execute_reply":"2026-03-31T16:27:28.609029Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. GENERATORS ","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=15,\n    zoom_range=0.15,\n    horizontal_flip=True\n)\n\nval_datagen = ImageDataGenerator(rescale=1./255)\n\ntrain_generator = train_datagen.flow_from_dataframe(\n    train_df,\n    directory=IMAGE_DIR,\n    x_col=\"id_code\",\n    y_col=\"diagnosis\",\n    target_size=(IMG_SIZE, IMG_SIZE),\n    batch_size=BATCH_SIZE,\n    class_mode=\"raw\",\n    shuffle=True\n)\n\nval_generator = val_datagen.flow_from_dataframe(\n    val_df,\n    directory=IMAGE_DIR,\n    x_col=\"id_code\",\n    y_col=\"diagnosis\",\n    target_size=(IMG_SIZE, IMG_SIZE),\n    batch_size=BATCH_SIZE,\n    class_mode=\"raw\",\n    shuffle=False\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T16:27:34.349569Z","iopub.execute_input":"2026-03-31T16:27:34.349839Z","iopub.status.idle":"2026-03-31T16:27:39.761896Z","shell.execute_reply.started":"2026-03-31T16:27:34.349815Z","shell.execute_reply":"2026-03-31T16:27:39.761073Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6. MODEL (EfficientNet + Regression)","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.applications import EfficientNetB4\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Dropout, BatchNormalization\nfrom tensorflow.keras.models import Model\n\nbase_model = EfficientNetB4(\n    weights='imagenet',\n    include_top=False,\n    input_shape=(IMG_SIZE, IMG_SIZE, 3)\n)\n\n# FULL UNFREEZE \nfor layer in base_model.layers:\n    layer.trainable = True\n\nx = base_model.output\nx = GlobalAveragePooling2D()(x)\n\nx = Dense(128, activation='relu')(x)\nx = BatchNormalization()(x)\nx = Dropout(0.2)(x)\n\noutput = Dense(1, activation='sigmoid')(x)\n\nmodel = Model(inputs=base_model.input, outputs=output)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T16:27:44.605753Z","iopub.execute_input":"2026-03-31T16:27:44.606295Z","iopub.status.idle":"2026-03-31T16:27:50.891390Z","shell.execute_reply.started":"2026-03-31T16:27:44.606265Z","shell.execute_reply":"2026-03-31T16:27:50.890649Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 7. COMPILE","metadata":{}},{"cell_type":"code","source":"model.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=2e-4),  \n    loss=tf.keras.losses.Huber(delta=1.0),                   \n    metrics=['mae']\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T16:28:25.861982Z","iopub.execute_input":"2026-03-31T16:28:25.862763Z","iopub.status.idle":"2026-03-31T16:28:25.878618Z","shell.execute_reply.started":"2026-03-31T16:28:25.862730Z","shell.execute_reply":"2026-03-31T16:28:25.878044Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n## 8. QWK FUNCTION","metadata":{}},{"cell_type":"code","source":"def compute_qwk(y_true, y_pred):\n    y_true = y_true * 4\n    y_pred = y_pred * 4\n\n    y_pred = np.clip(y_pred, 0, 4)\n    y_pred = np.round(y_pred).astype(int)\n\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T16:28:53.049743Z","iopub.execute_input":"2026-03-31T16:28:53.050055Z","iopub.status.idle":"2026-03-31T16:28:53.054975Z","shell.execute_reply.started":"2026-03-31T16:28:53.050027Z","shell.execute_reply":"2026-03-31T16:28:53.054305Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class QWKCallback(tf.keras.callbacks.Callback):\n    def on_epoch_end(self, epoch, logs=None):\n        preds = self.model.predict(val_generator).flatten()\n        y_true = val_df['diagnosis'].values\n        \n        qwk = compute_qwk(y_true, preds)\n        print(f\"\\nEpoch {epoch+1} QWK: {qwk:.4f}\")\n        \n        print(\"Pred range:\", preds.min(), preds.max())\n\n\ncheckpoint = tf.keras.callbacks.ModelCheckpoint(\n    \"best_model.h5\",\n    monitor=\"val_loss\",\n    save_best_only=True,\n    verbose=1\n)\n\nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(\n    monitor='val_loss',\n    factor=0.5,\n    patience=1,\n    min_lr=1e-6,\n    verbose=1\n)\n\nearly_stop = tf.keras.callbacks.EarlyStopping(\n    monitor='val_loss',\n    patience=4,\n    restore_best_weights=True,\n    verbose=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T16:32:20.844870Z","iopub.execute_input":"2026-03-31T16:32:20.845244Z","iopub.status.idle":"2026-03-31T16:32:20.852246Z","shell.execute_reply.started":"2026-03-31T16:32:20.845213Z","shell.execute_reply":"2026-03-31T16:32:20.851466Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 9. TRAIN","metadata":{}},{"cell_type":"code","source":"history = model.fit(\n    train_generator,\n    validation_data=val_generator,\n    epochs=EPOCHS,\n    callbacks=[QWKCallback(), checkpoint, reduce_lr, early_stop],\n    verbose=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T16:32:25.755741Z","iopub.execute_input":"2026-03-31T16:32:25.756344Z","iopub.status.idle":"2026-03-31T18:33:07.617250Z","shell.execute_reply.started":"2026-03-31T16:32:25.756315Z","shell.execute_reply":"2026-03-31T18:33:07.616468Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 10. THRESHOLD OPTIMIZATION","metadata":{}},{"cell_type":"code","source":"def qwk_loss(thresholds, preds, labels):\n    preds = preds * 4\n    labels = labels * 4\n    \n    preds = np.digitize(preds, thresholds)\n    \n    return -cohen_kappa_score(labels, preds, weights='quadratic')\n\n\ndef optimize_thresholds(preds, labels):\n    initial = [0.5, 1.5, 2.5, 3.5]\n    \n    result = minimize(\n        qwk_loss,\n        initial,\n        args=(preds, labels),\n        method='nelder-mead'\n    )\n    \n    return result.x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T18:37:55.123700Z","iopub.execute_input":"2026-03-31T18:37:55.124008Z","iopub.status.idle":"2026-03-31T18:37:55.129115Z","shell.execute_reply.started":"2026-03-31T18:37:55.123978Z","shell.execute_reply":"2026-03-31T18:37:55.128377Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 11. FINAL EVALUATION","metadata":{}},{"cell_type":"code","source":"preds = model.predict(val_generator).flatten()\n\nbest_thresh = optimize_thresholds(preds, val_df['diagnosis'].values)\n\nprint(\"Best thresholds:\", best_thresh)\n\nfinal_preds = np.digitize(preds * 4, best_thresh)\n\nqwk = cohen_kappa_score(\n    val_df['diagnosis'].values * 4,\n    final_preds,\n    weights='quadratic'\n)\n\nprint(\"FINAL QWK:\", qwk)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T18:37:59.606926Z","iopub.execute_input":"2026-03-31T18:37:59.607200Z","iopub.status.idle":"2026-03-31T18:39:07.391907Z","shell.execute_reply.started":"2026-03-31T18:37:59.607174Z","shell.execute_reply":"2026-03-31T18:39:07.391054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}