{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":4104,"databundleVersionId":46661,"sourceType":"competition"},{"sourceId":7866129,"sourceType":"datasetVersion","datasetId":4614938},{"sourceId":7869237,"sourceType":"datasetVersion","datasetId":4617269}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#!pip install -q keras-cv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T05:59:27.567751Z","iopub.execute_input":"2025-06-01T05:59:27.568570Z","iopub.status.idle":"2025-06-01T05:59:36.185256Z","shell.execute_reply.started":"2025-06-01T05:59:27.568542Z","shell.execute_reply":"2025-06-01T05:59:36.184189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom glob import glob\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.applications import ConvNeXtTiny\nfrom tensorflow.keras import layers, models, callbacks\nfrom tensorflow.keras import mixed_precision","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T06:12:42.567860Z","iopub.execute_input":"2025-06-01T06:12:42.568188Z","iopub.status.idle":"2025-06-01T06:12:42.572891Z","shell.execute_reply.started":"2025-06-01T06:12:42.568166Z","shell.execute_reply":"2025-06-01T06:12:42.572093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BATCH_SIZE = 32\nIMG_SIZE = (224, 224)\nAUTOTUNE = tf.data.AUTOTUNE\n\npolicy = mixed_precision.Policy('mixed_float16')\nmixed_precision.set_global_policy(policy)\n\ntrain_csv_path = \"/kaggle/input/diabetic-retinopathy-detection/trainLabels.csv.zip\"\ntrain_images_dir = \"/kaggle/input/diabetic-retinopathy-train-unzipped/train/\"\ntest_images_dir = \"/kaggle/input/diabetic-retinopathy-test-unzipped/test/\"\nsubmission_csv_path = \"/kaggle/input/diabetic-retinopathy-detection/sampleSubmission.csv.zip\"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv(train_csv_path)\ndf_train[\"filepath\"] = df_train[\"image\"].apply(lambda x: os.path.join(train_images_dir, f\"{x}.jpeg\"))\n\ntrain_df, val_df = train_test_split(df_train, test_size=0.2, stratify=df_train[\"level\"], random_state=42)\n\ndf_submission = pd.read_csv(submission_csv_path)\ndf_submission[\"filepath\"] = df_submission[\"image\"].apply(lambda x: os.path.join(test_images_dir, f\"{x}.jpeg\"))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_image(path, label=None):\n    img = tf.io.read_file(path)\n    img = tf.image.decode_png(img, channels=3)\n    img = tf.image.resize(img, IMG_SIZE)\n    img = tf.cast(img, tf.float32) / 255.0\n    return (img, label) if label is not None else img","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ds = tf.data.Dataset.from_tensor_slices((train_df[\"filepath\"], train_df[\"level\"]))\nval_ds = tf.data.Dataset.from_tensor_slices((val_df[\"filepath\"], val_df[\"level\"]))\ntrain_ds = train_ds.shuffle(1024).map(load_image, AUTOTUNE).batch(BATCH_SIZE).prefetch(AUTOTUNE)\nval_ds = val_ds.map(load_image, AUTOTUNE).batch(BATCH_SIZE).prefetch(AUTOTUNE)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"base_model = ConvNeXtTiny(include_top=False, input_shape=(*IMG_SIZE, 3), weights=\"imagenet\")\nbase_model.trainable = True\n\nmodel = models.Sequential([\n    layers.Input(shape=(*IMG_SIZE, 3)),\n    base_model,\n    layers.GlobalAveragePooling2D(),\n    layers.Dropout(0.3),\n    layers.Dense(5, activation=\"softmax\")\n])\n\nmodel.compile(optimizer=\"adam\",\n              loss=\"sparse_categorical_crossentropy\",\n              metrics=[\"accuracy\"])\n\n# CALLBACKS\ncb = [\n    callbacks.ReduceLROnPlateau(factor=0.5, patience=2, verbose=1),\n    callbacks.EarlyStopping(patience=4, restore_best_weights=True)\n]\n\n# TRAIN\nmodel.fit(train_ds, validation_data=val_ds, epochs=15, callbacks=cb)\n\n# TEST\ntest_paths = sorted(glob('/kaggle/input/diabetic-retinopathy-test-unzipped/test/*.jpeg'))\ntest_df = pd.DataFrame({\n    'image': [os.path.basename(p).split('.')[0] for p in test_paths],\n    'filepath': test_paths\n})\n\n\ntest_ds = tf.data.Dataset.from_tensor_slices(df_submission[\"filepath\"].values)\ntest_ds = test_ds.map(load_image, AUTOTUNE).batch(BATCH_SIZE)\n\n# PREDICT\npreds = model.predict(test_ds, verbose=0)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_submission[\"level\"] = np.argmax(preds, axis=1)\ndf_submission[[\"image\", \"level\"]].to_csv(\"submission.csv\", index=False)\nprint(\"submission.csv saved\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}