{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":4104,"databundleVersionId":46661,"sourceType":"competition"},{"sourceId":952963,"sourceType":"datasetVersion","datasetId":505422}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Basic libraries\nimport os\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport random\nimport cv2\n\n# TensorFlow and Keras\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.models import Model, Sequential\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau, ModelCheckpoint\nfrom sklearn.utils import class_weight\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-21T15:54:05.571291Z","iopub.execute_input":"2025-08-21T15:54:05.571843Z","iopub.status.idle":"2025-08-21T15:54:05.576420Z","shell.execute_reply.started":"2025-08-21T15:54:05.571818Z","shell.execute_reply":"2025-08-21T15:54:05.575665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 2. Set directories and labels\ndata_dir = \"/kaggle/input/diabetic-retinopathy-224x224-2019-data/colored_images\"\nlabels = ['No_DR', 'Mild', 'Moderate', 'Severe', 'Proliferate_DR']\nlabel_to_int = {label: idx for idx, label in enumerate(labels)}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T15:34:41.300061Z","iopub.execute_input":"2025-08-21T15:34:41.300370Z","iopub.status.idle":"2025-08-21T15:34:41.304690Z","shell.execute_reply.started":"2025-08-21T15:34:41.300348Z","shell.execute_reply":"2025-08-21T15:34:41.303898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count images per class\nclass_counts = {}\nfor cls in os.listdir(data_dir):  # <- use os.listdir here\n    cls_path = os.path.join(data_dir, cls)\n    count = len(os.listdir(cls_path))\n    class_counts[cls] = count\n\nprint(\"Class distribution:\")\nfor cls, count in class_counts.items():\n    print(f\"{cls}: {count}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T15:49:50.782323Z","iopub.execute_input":"2025-08-21T15:49:50.782620Z","iopub.status.idle":"2025-08-21T15:49:50.797545Z","shell.execute_reply.started":"2025-08-21T15:49:50.782599Z","shell.execute_reply":"2025-08-21T15:49:50.796786Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 3. Preview images using cv2\nplt.figure(figsize=(20, 12))\nimages_per_label = 5\nfor i, label in enumerate(labels):\n    folder_path = os.path.join(data_dir, label)\n    for j in range(images_per_label):\n        img_file = os.listdir(folder_path)[j]\n        img_path = os.path.join(folder_path, img_file)\n        img = cv2.imread(img_path)                 # BGR format\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)  # convert to RGB\n        plt.subplot(len(labels), images_per_label, i*images_per_label + j + 1)\n        plt.imshow(img)\n        plt.title(label)\n        plt.axis('off')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T15:36:14.081731Z","iopub.execute_input":"2025-08-21T15:36:14.082070Z","iopub.status.idle":"2025-08-21T15:36:15.725896Z","shell.execute_reply.started":"2025-08-21T15:36:14.082049Z","shell.execute_reply":"2025-08-21T15:36:15.725094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# 4. Compute class weights\n# =========================\nall_labels = []\nfor idx, cls in enumerate(labels):\n    all_labels.extend([idx]*class_counts[cls])\nall_labels = np.array(all_labels)\n\nweights = class_weight.compute_class_weight(\n    class_weight='balanced',\n    classes=np.unique(all_labels),\n    y=all_labels\n)\nclass_weights = dict(enumerate(weights))\nprint(\"Class weights:\", class_weights)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T15:54:10.201353Z","iopub.execute_input":"2025-08-21T15:54:10.201838Z","iopub.status.idle":"2025-08-21T15:54:10.209139Z","shell.execute_reply.started":"2025-08-21T15:54:10.201814Z","shell.execute_reply":"2025-08-21T15:54:10.208426Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 4. Create ImageDataGenerator with augmentation\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    vertical_flip=True,\n    validation_split=0.2\n)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T15:55:27.057407Z","iopub.execute_input":"2025-08-21T15:55:27.058006Z","iopub.status.idle":"2025-08-21T15:55:27.062048Z","shell.execute_reply.started":"2025-08-21T15:55:27.057972Z","shell.execute_reply":"2025-08-21T15:55:27.061413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Validation generator (no augmentation)\nval_datagen = ImageDataGenerator(rescale=1./255, validation_split=0.2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T15:55:40.170551Z","iopub.execute_input":"2025-08-21T15:55:40.170830Z","iopub.status.idle":"2025-08-21T15:55:40.174424Z","shell.execute_reply.started":"2025-08-21T15:55:40.170810Z","shell.execute_reply":"2025-08-21T15:55:40.173784Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Training generator\ntrain_generator = train_datagen.flow_from_directory(\n    data_dir,\n    target_size=(224,224),\n    batch_size=16,\n    class_mode='categorical',\n    subset='training',\n    shuffle=True\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T15:55:51.243177Z","iopub.execute_input":"2025-08-21T15:55:51.243466Z","iopub.status.idle":"2025-08-21T15:55:53.892003Z","shell.execute_reply.started":"2025-08-21T15:55:51.243445Z","shell.execute_reply":"2025-08-21T15:55:53.891283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Validation generator\nval_generator = train_datagen.flow_from_directory(\n    data_dir,\n    target_size=(224,224),\n    batch_size=16,\n    class_mode='categorical',\n    subset='validation',\n    shuffle=False\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T15:56:13.210761Z","iopub.execute_input":"2025-08-21T15:56:13.211717Z","iopub.status.idle":"2025-08-21T15:56:13.276035Z","shell.execute_reply.started":"2025-08-21T15:56:13.211687Z","shell.execute_reply":"2025-08-21T15:56:13.275519Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load ResNet50 without top layers\nbase_model = ResNet50(weights='imagenet', include_top=False, input_shape=(224,224,3))\n# Freeze base layers\nfor layer in base_model.layers:\n    layer.trainable = False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T15:56:23.412303Z","iopub.execute_input":"2025-08-21T15:56:23.412782Z","iopub.status.idle":"2025-08-21T15:56:24.435613Z","shell.execute_reply.started":"2025-08-21T15:56:23.412760Z","shell.execute_reply":"2025-08-21T15:56:24.434815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Build Sequential model\nmodel = Sequential([\n    base_model,\n    GlobalAveragePooling2D(),\n    Dense(512, activation='relu'),\n    Dense(len(labels), activation='softmax')  # 5 classes\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T15:56:36.966371Z","iopub.execute_input":"2025-08-21T15:56:36.966666Z","iopub.status.idle":"2025-08-21T15:56:36.989857Z","shell.execute_reply.started":"2025-08-21T15:56:36.966644Z","shell.execute_reply":"2025-08-21T15:56:36.989056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Compile\nmodel.compile(optimizer=Adam(learning_rate=1e-5),\n              loss='categorical_crossentropy',\n              metrics=['accuracy'])\n\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T15:57:21.894446Z","iopub.execute_input":"2025-08-21T15:57:21.895019Z","iopub.status.idle":"2025-08-21T15:57:21.918226Z","shell.execute_reply.started":"2025-08-21T15:57:21.894994Z","shell.execute_reply":"2025-08-21T15:57:21.917520Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = model.fit(\n    train_generator,\n    validation_data=val_generator,\n    epochs=10,\n    class_weight=class_weights\n\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T15:57:25.278178Z","iopub.execute_input":"2025-08-21T15:57:25.278705Z","iopub.status.idle":"2025-08-21T16:04:58.005895Z","shell.execute_reply.started":"2025-08-21T15:57:25.278680Z","shell.execute_reply":"2025-08-21T16:04:58.005317Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}