{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"}],"dockerImageVersionId":31013,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"pip install packages","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip uninstall -y scikit-learn imbalanced-learn numpy\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install --upgrade imbalanced-learn\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"  import sklearn\n  import imblearn\n  print(sklearn.__version__)\n  print(imblearn.__version__)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"  !pip install -U scikit-learn==0.24.0 imbalanced-learn\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from imblearn.over_sampling import SMOTE\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip uninstall -y imbalanced-learn scikit-learn\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install imbalanced-learn==0.10.1 scikit-learn==0.24.2\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install scikit-learn==1.5\n!pip install imbalanced-learn==0.12.4\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sklearn\nimport imblearn\n\nprint(\"scikit-learn version:\", sklearn.__version__)\nprint(\"imbalanced-learn version:\", imblearn.__version__)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install scikit-learn==1.5\n!pip install imbalanced-learn==0.13.0\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip uninstall -y scikit-learn imbalanced-learn\n#after doing this do the next\n!pip install -U scikit-learn==1.3.2\n!pip install -U imbalanced-learn==0.11.0\n#after doing this click on restart and clear cell output\nfrom imblearn.over_sampling import SMOTE\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -U scikit-learn==1.3.2\n!pip install -U imbalanced-learn==0.11.0\n#after doing this click on restart and clear cell output","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from imblearn.over_sampling import SMOTE\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T05:14:42.838222Z","iopub.execute_input":"2025-04-28T05:14:42.838505Z","iopub.status.idle":"2025-04-28T05:14:43.522577Z","shell.execute_reply.started":"2025-04-28T05:14:42.838483Z","shell.execute_reply":"2025-04-28T05:14:43.521781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ------------------------------------------\n# Step 1: Generate SMOTE Images and CSV\n# ------------------------------------------\n\nimport os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport cv2\nimport math\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator, array_to_img\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\nfrom imblearn.over_sampling import SMOTE\nfrom tensorflow.keras.applications import ResNet50, InceptionV3\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\nfrom tensorflow.keras.applications import ResNet50, InceptionV3, MobileNetV3Large, EfficientNetV2S\n\nfrom tensorflow.keras import mixed_precision\nmixed_precision.set_global_policy('mixed_float16')\n\n# Paths\nbase_path = '/kaggle/working/'\ntrain_csv_path = '/kaggle/input/aptos2019-blindness-detection/train.csv'\ntrain_images_path = '/kaggle/input/aptos2019-blindness-detection/train_images'\n\n# Directory to save generated images\nsave_dir = os.path.join(base_path, 'generated_classes_1_2_3_4')\nos.makedirs(save_dir, exist_ok=True)\n\n# Load CSV\ntrain_df = pd.read_csv(train_csv_path)\n\n# Target classes\ntarget_classes = [1, 2, 3, 4]\nfiltered_df = train_df[train_df['diagnosis'].isin(target_classes)]\n\nprint(f\"Number of images in classes 1, 2, 3, 4: {len(filtered_df)}\")\n\n# Prepare temp directory for flow_from_directory\ntemp_dir = os.path.join(base_path, 'temp_classes_1_2_3_4')\nfor cls in target_classes:\n    os.makedirs(os.path.join(temp_dir, str(cls)), exist_ok=True)\n\n# Copy images\nfor idx, row in filtered_df.iterrows():\n    img_id = row['id_code']\n    label = row['diagnosis']\n    src_path = os.path.join(train_images_path, img_id + '.png')\n    dst_path = os.path.join(temp_dir, str(label), img_id + '.png')\n    if os.path.exists(src_path):\n        os.system(f'cp \"{src_path}\" \"{dst_path}\"')\n\n# ImageDataGenerator\nimagegen = ImageDataGenerator(rescale=1./255)\n\ntrain_generator = imagegen.flow_from_directory(\n    temp_dir,\n    class_mode=\"categorical\",\n    shuffle=False,\n    batch_size=128,\n    target_size=(512, 512),\n    seed=42\n)\n\n# Load all images\nx = np.concatenate([next(train_generator)[0] for _ in range(train_generator.__len__())])\ny = np.concatenate([next(train_generator)[1] for _ in range(train_generator.__len__())])\n\nprint(f\"x shape: {x.shape}\")\nprint(f\"y shape: {y.shape}\")\n\n# Labels\ny_labels = np.argmax(y, axis=1)\n\n# Correct label mapping (since flow_from_directory assigns labels 0,1,2,3 for folders 1,2,3,4)\nmapping = {0:1, 1:2, 2:3, 3:4}\ny_labels = np.vectorize(mapping.get)(y_labels)\n\n# Keep only target classes 1,2,3,4\nmask = np.isin(y_labels, target_classes)\nx = x[mask]\ny = y[mask]\ny_labels = y_labels[mask]\n\n# NOW Flatten images for SMOTE\nX_train = x.reshape(x.shape[0], -1)\n\n# Count existing images per class\nexisting_counts = {cls: np.sum(y_labels == cls) for cls in target_classes}\nprint(\"Existing counts per class:\", existing_counts)\n\n# Find maximum number of images among the 4 target classes\nmax_count = max(existing_counts.values())\nprint(f\"Max count (target for all classes): {max_count}\")\n\n# Extra needed per class\nextra_needed = {cls: max_count - existing_counts.get(cls, 0) for cls in target_classes}\nprint(f\"Extra images needed per class: {extra_needed}\")\n\n\n# Apply SMOTE\nsm = SMOTE(random_state=42)\nX_smote, y_smote = sm.fit_resample(X_train, y_labels)\n\nprint(f\"Shape after SMOTE - X: {X_smote.shape}, y: {y_smote.shape}\")\n\n# Store original number of images\nnum_original = x.shape[0]\n\n# Collect generated samples\nXsmote_img_list = []\nys_smote_list = []\nclass_counts = {cls: 0 for cls in target_classes}\n\nfor idx in range(num_original, len(X_smote)):\n    label = y_smote[idx]\n    if label in target_classes and class_counts[label] < extra_needed[label]:\n        Xsmote_img_list.append(X_smote[idx])\n        ys_smote_list.append(label)\n        class_counts[label] += 1\n    if all(class_counts[c] >= extra_needed[c] for c in target_classes):\n        break\n\nXsmote_img_array = np.array(Xsmote_img_list).reshape(-1, 512, 512, 3)\nys_smote_array = np.array(ys_smote_list)\n\nprint(f\"Generated {len(Xsmote_img_array)} synthetic images.\")\n\n# Save generated images\nfor cls in target_classes:\n    os.makedirs(os.path.join(save_dir, str(cls)), exist_ok=True)\n\n# Save images and prepare CSV entries\nrecords = []\n\nfor i in range(len(Xsmote_img_array)):\n    label = ys_smote_array[i]\n    img_name = f'smote_{label}_{i}.png'\n    pil_img = array_to_img(Xsmote_img_array[i] * 255.0)  # Unnormalize\n    img_save_path = os.path.join(save_dir, str(label), img_name)\n    pil_img.save(img_save_path)\n    records.append((img_name, label))\n\n\n# Save CSV file\ncsv_df = pd.DataFrame(records, columns=[\"filename\", \"label\"])\ncsv_path = os.path.join(base_path, \"generated_images_labels.csv\")\ncsv_df.to_csv(csv_path, index=False)\n\nprint(f\"CSV file saved to: {csv_path}\")\n\n# Plot 2 generated images per class\nplt.figure(figsize=(12, 8))\nshown_per_class = {cls: 0 for cls in target_classes}\nsubplot_idx = 1\n\nfor idx in range(len(Xsmote_img_array)):\n    label = ys_smote_array[idx]\n    if shown_per_class[label] < 2:\n        plt.subplot(4, 2, subplot_idx)\n        plt.imshow(Xsmote_img_array[idx])\n        plt.title(f'Class {label}')\n        plt.axis('off')\n        shown_per_class[label] += 1\n        subplot_idx += 1\n    if subplot_idx > 8:\n        break\n\nplt.suptitle('2 Generated Images from Each Class', fontsize=20)\nplt.tight_layout()\nplt.show()","metadata":{"_uuid":"15ec4262-c390-4c68-ab8c-f94a1ec5c58f","_cell_guid":"db9ad09f-63df-427f-8301-30dc0a6277b2","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ------------------------------------------\n# Step 2: Preprocessing Both Original and SMOTE Images\n# ------------------------------------------\n\nIMG_SIZE = 224\nNB_CHANNELS = 3\n\ndef get_pad_width(im, new_shape, is_rgb=True):\n    pad_diff = new_shape - im.shape[0], new_shape - im.shape[1]\n    t, b = math.floor(pad_diff[0]/2), math.ceil(pad_diff[0]/2)\n    l, r = math.floor(pad_diff[1]/2), math.ceil(pad_diff[1]/2)\n    return ((t,b), (l,r), (0,0)) if is_rgb else ((t,b), (l,r))\n\ndef standardize(x):\n    x = x.astype(np.float32)\n    x = x / np.max(x)\n    return (x - np.mean(x)) / (np.std(x))\n\ndef normalize(img):\n    img = ((img - np.min(img)) / (np.max(img) - np.min(img))) * 255\n    return img.astype(np.uint8)\n\ndef crop_image(img, tol=10):\n    def crop_image_1(img):\n        mask = img > tol\n        return img[np.ix_(mask.any(1), mask.any(0))]\n\n    if img.ndim == 2:\n        return crop_image_1(img)\n    elif img.ndim == 3:\n        try:\n            img_cpy = img.copy()\n            h, w, _ = img.shape\n            img1 = cv2.resize(crop_image_1(img[:,:,0]), (w,h))\n            img2 = cv2.resize(crop_image_1(img[:,:,1]), (w,h))\n            img3 = cv2.resize(crop_image_1(img[:,:,2]), (w,h))\n            img[:,:,0] = img1\n            img[:,:,1] = img2\n            img[:,:,2] = img3\n            return img\n        except:\n            return img_cpy\n\ndef preprocess_image(img_path):\n    im = cv2.imread(img_path)\n    if im is None:\n        return None\n    im = cv2.cvtColor(im, cv2.COLOR_BGR2RGB)\n    im = normalize(im)\n    im = crop_image(im)\n    im = cv2.resize(im, (IMG_SIZE, IMG_SIZE))\n    im_lab = cv2.cvtColor(im, cv2.COLOR_RGB2LAB)\n    l_channel, a_channel, b_channel = cv2.split(im_lab)\n    clahe = cv2.createCLAHE(clipLimit=0.1, tileGridSize=(2, 2))\n    l_channel = clahe.apply(l_channel)\n    im_lab = cv2.merge([l_channel, a_channel, b_channel])\n    im = cv2.cvtColor(im_lab, cv2.COLOR_LAB2RGB)\n    im = cv2.addWeighted(im, 4, cv2.GaussianBlur(im, (0, 0), IMG_SIZE/10), -4, 128)\n\n    mask = np.zeros((IMG_SIZE, IMG_SIZE), dtype=np.uint8)\n    cv2.circle(mask, (IMG_SIZE//2, IMG_SIZE//2), IMG_SIZE//2, 255, -1)\n    for c in range(3):\n        im[:,:,c] = np.where(mask==255, im[:,:,c], 0)\n\n    return im.astype(np.uint8)\n\n\n#c\ndef augment_image(img):\n    datagen = tf.keras.preprocessing.image.ImageDataGenerator(\n        rotation_range=20,\n        horizontal_flip=True\n    )\n    img = img.reshape((1,) + img.shape)\n    return next(datagen.flow(img, batch_size=1))[0].astype(np.uint8).squeeze()\n#c\n\n#c\ntrain_df.rename(columns={'id_code': 'filename', 'diagnosis': 'label'})\ntrain_df['filename'] = train_df['id_code'] + '.png'\ntrain_df.rename(columns={'diagnosis': 'label'}, inplace=True)\n#c\n# Combine original and generated dataframes\noriginal_df = train_df.rename(columns={'filename': 'filename', 'label': 'label'})\ngenerated_df = pd.read_csv(csv_path)\nfinal_df = pd.concat([original_df, generated_df], ignore_index=True)\n\n# Preprocess and save\nprocessed_dir = os.path.join(base_path, 'final_processed_images')\nos.makedirs(processed_dir, exist_ok=True)\n\n#c\n\nnew_records = []\n\nfor idx, row in train_df.iterrows():\n    img_name = row['filename']\n    label = row['label']\n    img_path = os.path.join(train_images_path, img_name)     #img_path = os.path.join(train_images_path, img_name)\n    img = preprocess_image(img_path)\n\n    new_name = f\"{img_name.split('.')[0]}_original.png\"\n    cv2.imwrite(os.path.join(processed_dir, new_name),img)\n\n    new_records.append({'filename': new_name, 'label': label})    #new_records.append({'filename': new_name.replace('.jpg', '.png'), 'label': label})\n\n\nfor idx, row in final_df.iterrows():\n    img_name = row['filename']\n    label = row['label']\n\n    # Determine if image is SMOTE or original\n    if img_name.startswith('smote_'):\n        img_path = os.path.join(save_dir, str(label), img_name)\n        img = preprocess_image(img_path)\n\n        new_name = f\"{img_name.split('.')[0]}_aug{i}.png\"\n        cv2.imwrite(os.path.join(processed_dir, new_name), img)\n\n        new_records.append({'filename': new_name, 'label': label})","metadata":{"_uuid":"15ec4262-c390-4c68-ab8c-f94a1ec5c58f","_cell_guid":"db9ad09f-63df-427f-8301-30dc0a6277b2","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_processed_df = pd.DataFrame(new_records)\nfinal_processed_csv = os.path.join(base_path, 'final_processed_data.csv')\nfinal_processed_df.to_csv(final_processed_csv, index=False)\n\nimport matplotlib.pyplot as plt\nimport cv2\n\nplt.figure(figsize=(12, 8))\n\n# Define classes in desired order\ntarget_classes_int = [1, 2, 3, 4]\nshown_per_class = {cls: 0 for cls in target_classes_int}\nsubplot_idx = 1\n\nfor cls in target_classes_int:\n    count = 0\n    for record in new_records:\n        label = int(record['label'])\n        if label != cls:\n            continue\n\n        img_path = os.path.join(processed_dir, record['filename'])\n        img = cv2.imread(img_path)\n        if img is not None:\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n            plt.subplot(4, 2, subplot_idx)\n            plt.imshow(img)\n            plt.title(f'Class {label}')\n            plt.axis('off')\n            subplot_idx += 1\n            count += 1\n        if count >= 2:\n            break\n\nplt.suptitle('2 Preprocessed Images from Each Class (Grouped)', fontsize=20)\nplt.tight_layout()\nplt.show()","metadata":{"_uuid":"15ec4262-c390-4c68-ab8c-f94a1ec5c58f","_cell_guid":"db9ad09f-63df-427f-8301-30dc0a6277b2","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ------------------------------------------\n# Step 3: Split Data\n# ------------------------------------------\n\ntrain_df, test_val_df = train_test_split(final_processed_df, test_size=0.3, stratify=final_processed_df['label'], random_state=42)\nval_df, test_df = train_test_split(test_val_df, test_size=0.5, stratify=test_val_df['label'], random_state=42)\n\n# ------------------------------------------\n# Step 4: Model Training\n# ------------------------------------------\n\n# Convert labels to string\ntrain_df['label'] = train_df['label'].astype(str)\nval_df['label'] = val_df['label'].astype(str)\ntest_df['label'] = test_df['label'].astype(str)\n\n# ImageDataGenerators\ntrain_datagen = ImageDataGenerator(rescale=1./255)\nval_datagen = ImageDataGenerator(rescale=1./255)\n\n# Train, validation, and test generators\ntrain_generator = train_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    directory=processed_dir,\n    x_col='filename',\n    y_col='label',\n    target_size=(IMG_SIZE, IMG_SIZE),\n    batch_size=128,\n    class_mode='categorical',\n    shuffle=True,\n    seed=42\n)\n\nval_generator = val_datagen.flow_from_dataframe(\n    dataframe=val_df,\n    directory=processed_dir,\n    x_col='filename',\n    y_col='label',\n    target_size=(IMG_SIZE, IMG_SIZE),\n    batch_size=128,\n    class_mode='categorical',\n    shuffle=False\n)\n\ntest_generator = val_datagen.flow_from_dataframe(\n    dataframe=test_df,\n    directory=processed_dir,\n    x_col='filename',\n    y_col='label',\n    target_size=(IMG_SIZE, IMG_SIZE),\n    batch_size=128,\n    class_mode='categorical',\n    shuffle=False\n)\n\n\n# Model\nbase_model = EfficientNetV2S(include_top=False, weights='imagenet', input_shape=(IMG_SIZE, IMG_SIZE, 3))\nx = base_model.output\nx = GlobalAveragePooling2D()(x)\npredictions = Dense(5, activation='softmax')(x)\nmodel = Model(inputs=base_model.input, outputs=predictions)\n\n#base_model.trainable = False  # Freeze base for transfer learning\n\n\nmodel.compile(optimizer=Adam(), loss='categorical_crossentropy', metrics=['accuracy'])\n\n\n\n\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\n\n# Callbacks\nearly_stop = EarlyStopping(monitor='val_loss', patience=19, restore_best_weights=True, verbose=1)\n\ncheckpoint_path = os.path.join(base_path, 'best_model.keras')\nmodel_checkpoint = ModelCheckpoint(\n    filepath=checkpoint_path,\n    monitor='val_loss',\n    save_best_only=True,\n    save_weights_only=False,\n    verbose=1\n)\n\ncallbacks = [early_stop, model_checkpoint]\n\n# Training\nhistory = model.fit(\n    train_generator,\n    epochs=20,\n    validation_data=val_generator,\n    callbacks=callbacks\n)","metadata":{"_uuid":"15ec4262-c390-4c68-ab8c-f94a1ec5c58f","_cell_guid":"db9ad09f-63df-427f-8301-30dc0a6277b2","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save model\nmodel.save(os.path.join(base_path, 'InceptionV3_model.keras'))\n\n# Plot training curves\nplt.figure(figsize=(12,5))\nplt.subplot(1,2,1)\nplt.plot(history.history['accuracy'], label='Train Acc')\nplt.plot(history.history['val_accuracy'], label='Val Acc')\nplt.legend()\nplt.title('Accuracy')\n\nplt.subplot(1,2,2)\nplt.plot(history.history['loss'], label='Train Loss')\nplt.plot(history.history['val_loss'], label='Val Loss')\nplt.legend()\nplt.title('Loss')\n\nplt.show()\n\n# ------------------------------------------\n# Step 5: Testing\n# ------------------------------------------\n\ntest_preds = model.predict(test_generator)\ntest_preds_classes = np.argmax(test_preds, axis=1)\ntrue_classes = test_generator.classes\n\n# Confusion matrix\ncm = confusion_matrix(true_classes, test_preds_classes)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm)\ndisp.plot(cmap=plt.cm.Blues)\nplt.title('Confusion Matrix')\nplt.show()\n\n# Show 2 images per class with prediction\nplt.figure(figsize=(12, 8))\nclass_counts = {i: 0 for i in range(5)}\nsubplot_idx = 1\n\nfor i in range(len(test_generator.filenames)):\n    img = cv2.imread(os.path.join(processed_dir, test_generator.filenames[i]))\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    label = true_classes[i]\n    pred = test_preds_classes[i]\n\n    if class_counts[label] < 2:\n        plt.subplot(5, 2, subplot_idx)\n        plt.imshow(img)\n        plt.title(f\"True: {label} | Pred: {pred}\")\n        plt.axis('off')\n        class_counts[label] += 1\n        subplot_idx += 1\n\n    if subplot_idx > 10:\n        break\n\nplt.tight_layout()\nplt.show()","metadata":{"_uuid":"15ec4262-c390-4c68-ab8c-f94a1ec5c58f","_cell_guid":"db9ad09f-63df-427f-8301-30dc0a6277b2","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null}]}