{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-17T14:04:30.701628Z","iopub.execute_input":"2025-04-17T14:04:30.702963Z","iopub.status.idle":"2025-04-17T14:04:40.803681Z","shell.execute_reply.started":"2025-04-17T14:04:30.702917Z","shell.execute_reply":"2025-04-17T14:04:40.802301Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1. Import Necessary Libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport cv2\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense\nfrom tensorflow.keras.optimizers import Adam\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T14:04:40.806187Z","iopub.execute_input":"2025-04-17T14:04:40.806730Z","iopub.status.idle":"2025-04-17T14:05:00.491081Z","shell.execute_reply.started":"2025-04-17T14:04:40.806695Z","shell.execute_reply":"2025-04-17T14:05:00.489862Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Load and Visualize the Data","metadata":{}},{"cell_type":"code","source":"# Load the dataset\ndf = pd.read_csv('/kaggle/input/aptos2019-blindness-detection/train.csv')\n\n# Visualize class distribution\nplt.figure(figsize=(8, 4))\nsns.countplot(x='diagnosis', data=df)\nplt.title('Distribution of Diabetic Retinopathy Classes')\nplt.xlabel('Diagnosis')\nplt.ylabel('Count')\nplt.xticks(rotation=45)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T14:05:00.492225Z","iopub.execute_input":"2025-04-17T14:05:00.492832Z","iopub.status.idle":"2025-04-17T14:05:00.825945Z","shell.execute_reply.started":"2025-04-17T14:05:00.492803Z","shell.execute_reply":"2025-04-17T14:05:00.824601Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Preprocess Images","metadata":{}},{"cell_type":"code","source":"# Define image size\nIMG_SIZE = 224\n\n# Initialize lists to store images and labels\nX = []\ny = []\n\n# Loop through each row in the dataframe\nfor index, row in tqdm(df.iterrows(), total=df.shape[0]):\n    image_id = row['id_code']\n    label = row['diagnosis']\n    image_path = os.path.join('/kaggle/input/aptos2019-blindness-detection/train_images', f'{image_id}.png')\n    \n    # Read and preprocess the image\n    image = cv2.imread(image_path)\n    if image is not None:\n        image = cv2.resize(image, (IMG_SIZE, IMG_SIZE))\n        image = image / 255.0  # Normalize pixel values\n        X.append(image)\n        y.append(label)\n\n# Convert lists to NumPy arrays\nX = np.array(X)\ny = np.array(y)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T14:05:00.827184Z","iopub.execute_input":"2025-04-17T14:05:00.827579Z","iopub.status.idle":"2025-04-17T14:12:59.732367Z","shell.execute_reply.started":"2025-04-17T14:05:00.827546Z","shell.execute_reply":"2025-04-17T14:12:59.728188Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Split the Data","metadata":{}},{"cell_type":"code","source":"# Split into training and validation sets\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.15, stratify=y, random_state=42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T14:12:59.739765Z","iopub.execute_input":"2025-04-17T14:12:59.740920Z","iopub.status.idle":"2025-04-17T14:13:01.455224Z","shell.execute_reply.started":"2025-04-17T14:12:59.740751Z","shell.execute_reply":"2025-04-17T14:13:01.453285Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5. Build the CNN Model","metadata":{}},{"cell_type":"code","source":"# Initialize the model\nmodel = Sequential()\n\n# Add convolutional and pooling layers\nmodel.add(Conv2D(32, (3, 3), activation='relu', input_shape=(IMG_SIZE, IMG_SIZE, 3)))\nmodel.add(MaxPooling2D((2, 2)))\nmodel.add(Conv2D(64, (3, 3), activation='relu'))\nmodel.add(MaxPooling2D((2, 2)))\nmodel.add(Conv2D(128, (3, 3), activation='relu'))\nmodel.add(MaxPooling2D((2, 2)))\n\n# Flatten the output and add dense layers\nmodel.add(Flatten())\nmodel.add(Dense(128, activation='relu'))\nmodel.add(Dense(5, activation='softmax'))  # 5 classes for DR stages\n\n# Compile the model\nmodel.compile(optimizer=Adam(learning_rate=1e-4), loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T14:13:01.456708Z","iopub.execute_input":"2025-04-17T14:13:01.457038Z","iopub.status.idle":"2025-04-17T14:13:01.912334Z","shell.execute_reply.started":"2025-04-17T14:13:01.457017Z","shell.execute_reply":"2025-04-17T14:13:01.911452Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 6. Train the Model","metadata":{}},{"cell_type":"code","source":"# Train the model\nhistory = model.fit(\n    X_train, y_train,\n    validation_data=(X_val, y_val),\n    epochs=10,\n    batch_size=32\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T14:13:01.913459Z","iopub.execute_input":"2025-04-17T14:13:01.913765Z","iopub.status.idle":"2025-04-17T14:49:16.536295Z","shell.execute_reply.started":"2025-04-17T14:13:01.913744Z","shell.execute_reply":"2025-04-17T14:49:16.535206Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 7. Evaluate the Model","metadata":{}},{"cell_type":"code","source":"# Evaluate on validation set\nval_loss, val_accuracy = model.evaluate(X_val, y_val)\nprint(f'Validation Loss: {val_loss:.4f}')\nprint(f'Validation Accuracy: {val_accuracy:.4f}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T14:49:16.537857Z","iopub.execute_input":"2025-04-17T14:49:16.538147Z","iopub.status.idle":"2025-04-17T14:49:27.695013Z","shell.execute_reply.started":"2025-04-17T14:49:16.538124Z","shell.execute_reply":"2025-04-17T14:49:27.693597Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **1: Load and Preprocess Test Images**","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport cv2\nfrom tqdm import tqdm\n\n# Define image size\nIMG_SIZE = 224\n\n# Load the sample submission file\ntest_df = pd.read_csv('/kaggle/input/aptos2019-blindness-detection/sample_submission.csv')\n\n# Initialize list to store processed test images\nX_test = []\n\n# Loop through each image in the test set\nfor image_id in tqdm(test_df['id_code']):\n    image_path = os.path.join('/kaggle/input/aptos2019-blindness-detection/test_images', f'{image_id}.png')\n    image = cv2.imread(image_path)\n    if image is not None:\n        image = cv2.resize(image, (IMG_SIZE, IMG_SIZE))\n        image = image / 255.0  # Normalize pixel values\n        X_test.append(image)\n    else:\n        # If image is not found or cannot be read, append a zero array\n        X_test.append(np.zeros((IMG_SIZE, IMG_SIZE, 3)))\n\n# Convert list to NumPy array\nX_test = np.array(X_test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T14:52:25.145207Z","iopub.execute_input":"2025-04-17T14:52:25.145612Z","iopub.status.idle":"2025-04-17T14:53:38.465983Z","shell.execute_reply.started":"2025-04-17T14:52:25.145590Z","shell.execute_reply":"2025-04-17T14:53:38.464526Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#  2: Make Predictions","metadata":{}},{"cell_type":"code","source":"# Use the trained model to make predictions on the test set\npredictions = model.predict(X_test)\n\n# For each prediction, select the class with the highest probability\npredicted_classes = np.argmax(predictions, axis=1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T14:56:19.245515Z","iopub.execute_input":"2025-04-17T14:56:19.245907Z","iopub.status.idle":"2025-04-17T14:56:58.811099Z","shell.execute_reply.started":"2025-04-17T14:56:19.245879Z","shell.execute_reply":"2025-04-17T14:56:58.810138Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Verify Submission File","metadata":{}},{"cell_type":"code","source":"# Display the first few rows of the submission file\nprint(test_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T14:57:33.985654Z","iopub.execute_input":"2025-04-17T14:57:33.986607Z","iopub.status.idle":"2025-04-17T14:57:34.003560Z","shell.execute_reply.started":"2025-04-17T14:57:33.986571Z","shell.execute_reply":"2025-04-17T14:57:34.002464Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.save('dr_prediction.h5')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T15:06:13.656973Z","iopub.execute_input":"2025-04-17T15:06:13.657360Z","iopub.status.idle":"2025-04-17T15:06:14.130642Z","shell.execute_reply.started":"2025-04-17T15:06:13.657334Z","shell.execute_reply":"2025-04-17T15:06:14.129555Z"}},"outputs":[],"execution_count":null}]}