{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"}],"dockerImageVersionId":31011,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip uninstall -y scikit-learn imbalanced-learn\n#after doing this do the next","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -U scikit-learn==1.3.2\n!pip install -U imbalanced-learn==0.11.0\n#after doing this click on restart and clear cell output","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from imblearn.over_sampling import SMOTE\nfrom imblearn.combine import SMOTEENN\nfrom imblearn.over_sampling import ADASYN","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:23:42.743235Z","iopub.execute_input":"2025-04-30T20:23:42.743955Z","iopub.status.idle":"2025-04-30T20:23:43.378576Z","shell.execute_reply.started":"2025-04-30T20:23:42.743928Z","shell.execute_reply":"2025-04-30T20:23:43.377889Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport random\nimport shutil\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom tensorflow.keras.utils import to_categorical\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, confusion_matrix\nimport matplotlib.pyplot as plt\nfrom glob import glob\nimport gc\nfrom tensorflow.keras import mixed_precision\nimport math  # Side note: Added import for math, required by get_pad_width function\nfrom numpy import expand_dims\nfrom numpy import zeros\nfrom numpy import ones\nfrom numpy import vstack\nfrom numpy.random import randn\nfrom numpy.random import randint\nfrom keras.optimizers import Adam\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Reshape, Flatten, Conv2D, Conv2DTranspose, LeakyReLU, Dropout\nfrom tensorflow.keras.models import Sequential","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:23:52.227024Z","iopub.execute_input":"2025-04-30T20:23:52.227372Z","iopub.status.idle":"2025-04-30T20:24:09.521539Z","shell.execute_reply.started":"2025-04-30T20:23:52.227352Z","shell.execute_reply":"2025-04-30T20:24:09.520727Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"CSV_PATH = '/kaggle/input/aptos2019-blindness-detection/train.csv'\nIMAGE_DIR = '/kaggle/input/aptos2019-blindness-detection/train_images'\n#c\nimport gc\n#from tensorflow.keras import mixed_precision\n#mixed_precision.set_global_policy('mixed_float16')\n\n# =====================\n# STEP 1: Load Data\n# =====================\nSEED = 42\nrandom.seed(SEED)\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)\n\nCSV_PATH = '/kaggle/input/aptos2019-blindness-detection/train.csv'\nIMAGE_DIR = '/kaggle/input/aptos2019-blindness-detection/train_images'\n\ndf = pd.read_csv(CSV_PATH)\ndf['image'] = df['id_code'] + '.png'\ndf.rename(columns={'diagnosis': 'level'}, inplace=True)\n\n# Count total images\ntotal_images = len(df)\nprint(f\"Total images: {total_images}\")\n\n# Count images per class\nclass_counts = df['level'].value_counts().sort_index()\nprint(\"\\nImages per class:\")\nprint(class_counts)\n\n# =====================\n# Plot Histogram of Class Distribution\n# =====================\n\nplt.figure(figsize=(8, 6))\nplt.bar(class_counts.index, class_counts.values, color='skyblue', edgecolor='black')\nplt.xticks(class_counts.index)\nplt.xlabel('Class Label')\nplt.ylabel('Number of Images')\nplt.title('Histogram of Images per Class')\nplt.grid(axis='y', linestyle='--', alpha=0.7)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:24:09.522872Z","iopub.execute_input":"2025-04-30T20:24:09.523387Z","iopub.status.idle":"2025-04-30T20:24:09.829173Z","shell.execute_reply.started":"2025-04-30T20:24:09.523359Z","shell.execute_reply":"2025-04-30T20:24:09.828411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n#  Plot 2 Images from Each Class\n# =====================\nprint(\"Showing two raw images from each class\")\nplt.figure(figsize=(15, 10))\n\nfor class_label in range(5):  # Classes 0 to 4\n    class_images = df[df['level'] == class_label]['image'].values\n    selected_images = np.random.choice(class_images, 2, replace=False)\n    \n    for i, img_name in enumerate(selected_images):\n        img_path = os.path.join(IMAGE_DIR, img_name)\n        img = cv2.imread(img_path)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        \n        plt.subplot(5, 2, class_label * 2 + i + 1)\n        plt.imshow(img)\n        plt.title(f'Class {class_label}')\n        plt.axis('off')\n\nplt.tight_layout()\nplt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:24:09.829787Z","iopub.execute_input":"2025-04-30T20:24:09.829994Z","iopub.status.idle":"2025-04-30T20:24:16.593013Z","shell.execute_reply.started":"2025-04-30T20:24:09.829977Z","shell.execute_reply":"2025-04-30T20:24:16.592176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# STEP 2: Preprocessing + Augmentation\n# =====================\nIMG_SIZE = 224\nNB_CHANNELS = 3\n\ndef get_pad_width(im, new_shape, is_rgb=True):\n    pad_diff = new_shape - im.shape[0], new_shape - im.shape[1]\n    t, b = math.floor(pad_diff[0]/2), math.ceil(pad_diff[0]/2)\n    l, r = math.floor(pad_diff[1]/2), math.ceil(pad_diff[1]/2)\n    if is_rgb:\n        pad_width = ((t,b), (l,r), (0, 0))\n    else:\n        pad_width = ((t,b), (l,r))\n    return pad_width\n\ndef standardize(x):\n    x = x.astype(np.float32)\n    x = x / np.max(x)\n    return (x - np.mean(x)) / (np.std(x))\n\ndef normalize(img):\n    img = ((img - np.min(img)) / (np.max(img) - np.min(img))) * 255\n    return img.astype(np.uint8)\n\ndef crop_image(img, tol=10):\n    def crop_image_1(img):\n        mask = img > tol\n        return img[np.ix_(mask.any(1), mask.any(0))]\n    \n    if img.ndim == 2:\n        return crop_image_1(img)\n    \n    elif img.ndim == 3:\n        try:\n            img_cpy = img.copy()\n            h, w, _ = img.shape\n            img1 = cv2.resize(crop_image_1(img[:, :, 0]), (w, h))\n            img2 = cv2.resize(crop_image_1(img[:, :, 1]), (w, h))\n            img3 = cv2.resize(crop_image_1(img[:, :, 2]), (w, h))\n\n            img[:,:,0] = img1\n            img[:,:,1] = img2\n            img[:,:,2] = img3\n            return img\n        except:\n            return img_cpy\n\ndef preprocess_image(img_name, label=None, base_dir=IMAGE_DIR):\n    img_path = os.path.join(base_dir, img_name)\n    im = cv2.imread(img_path)\n    if im is None:\n        print(f\"Failed to load {img_path}\")\n        return None\n\n    im = cv2.cvtColor(im, cv2.COLOR_BGR2RGB)\n    im = normalize(im)\n    im = crop_image(im)\n    im = cv2.resize(im, (IMG_SIZE, IMG_SIZE))\n\n    \n\n    # Note: Applying CLAHE after resize to enhance contrast while preserving color\n    im_lab = cv2.cvtColor(im, cv2.COLOR_RGB2LAB)\n    l_channel, a_channel, b_channel = cv2.split(im_lab)\n    clahe = cv2.createCLAHE(clipLimit=0.1, tileGridSize=(2, 2))\n    l_channel = clahe.apply(l_channel)\n    im_lab = cv2.merge([l_channel, a_channel, b_channel])\n    im = cv2.cvtColor(im_lab, cv2.COLOR_LAB2RGB)\n    \n    im = cv2.addWeighted(im, 4, cv2.GaussianBlur(im, (0, 0), IMG_SIZE / 10), -4, 128)\n\n     # Mask background to black using circular ROI\n    mask = np.zeros((IMG_SIZE, IMG_SIZE), dtype=np.uint8)  # ← new\n    cv2.circle(mask, (IMG_SIZE // 2, IMG_SIZE // 2), IMG_SIZE // 2, 255, -1)  # ← new\n    for c in range(3):  # ← new\n        im[:, :, c] = np.where(mask == 255, im[:, :, c], 0)  # ← new\n\n   \n    \n    return im.astype(np.uint8)\n\ndef augment_image(img):\n    datagen = tf.keras.preprocessing.image.ImageDataGenerator(\n        rotation_range=20,\n        horizontal_flip=True\n    )\n    img = img.reshape((1,) + img.shape)\n    return next(datagen.flow(img, batch_size=1))[0].astype(np.uint8).squeeze()\n\nprint(\"Applying preprocessing and augmentation...\")\n\nPREPROCESSED_DIR = 'processed_images'\nos.makedirs(PREPROCESSED_DIR, exist_ok=True)\nupdated_rows = []\n\n##for idx, row in df.iterrows():\n##    img_name = row['image']\n##    label = row['level']\n##    img = preprocess_image(img_name, label, IMAGE_DIR)\n##    if img is None:\n##        continue\n##    for i in range(2):  # 2 augmentations per image\n##        aug_img = augment_image(img)\n##        new_name = f\"{row['image'].split('.')[0]}_aug{i}.png\"\n##        cv2.imwrite(os.path.join(PREPROCESSED_DIR, new_name), aug_img)\n##        updated_rows.append({'image': new_name, 'level': row['level']})\n\nfor idx, row in df.iterrows():\n    img_name = row['image']\n    label = row['level']\n    img = preprocess_image(img_name, label, IMAGE_DIR)\n    if img is not None:\n        new_name = row['image'].replace('.jpg', '.png').replace('.jpeg', '.png')\n        cv2.imwrite(os.path.join(PREPROCESSED_DIR, new_name), img)\n        updated_rows.append({'image': new_name, 'level': row['level']})\n\naug_df = pd.DataFrame(updated_rows)\n\nprint(\"Done...\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:24:16.594529Z","iopub.execute_input":"2025-04-30T20:24:16.594769Z","iopub.status.idle":"2025-04-30T20:40:24.543599Z","shell.execute_reply.started":"2025-04-30T20:24:16.594751Z","shell.execute_reply":"2025-04-30T20:40:24.542722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count total images\ntotal_images = len(aug_df)\nprint(f\"Total images: {total_images}\")\n\n# Count images per class\nclass_counts = aug_df['level'].value_counts().sort_index()\nprint(\"\\nImages per class:\")\nprint(class_counts)\n\n\n# =====================\n# Plot Histogram of Class Distribution\n# =====================\n\nplt.figure(figsize=(8, 6))\nplt.bar(class_counts.index, class_counts.values, color='skyblue', edgecolor='black')\nplt.xticks(class_counts.index)\nplt.xlabel('Class Label')\nplt.ylabel('Number of Images')\nplt.title('Images per class after preprocessing and augmentation')\nplt.grid(axis='y', linestyle='--', alpha=0.7)\nplt.show()\n\n# =====================\n#  Plot 2 Images from Each Class\n# =====================\nprint(\"\\nShowing 2 preprocessed images from each class:\")\nplt.figure(figsize=(15, 10))\n\nfor class_label in range(5):  # Classes 0 to 4\n    class_images = aug_df[aug_df['level'] == class_label]['image'].values\n    selected_images = np.random.choice(class_images, 2, replace=False)\n    \n    for i, img_name in enumerate(selected_images):\n        img_path = os.path.join(PREPROCESSED_DIR, img_name)\n        img = cv2.imread(img_path)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        \n        plt.subplot(5, 2, class_label * 2 + i + 1)\n        plt.imshow(img)\n        plt.title(f'Class {class_label}')\n        plt.axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:40:24.544522Z","iopub.execute_input":"2025-04-30T20:40:24.545359Z","iopub.status.idle":"2025-04-30T20:40:25.320176Z","shell.execute_reply.started":"2025-04-30T20:40:24.545334Z","shell.execute_reply":"2025-04-30T20:40:25.319464Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ignore this cell\n\n# =====================\n#  Show first 10 preprocessed images from \"PREPROCESSED_DIR\"\n# =====================\n#print(\"\\nShowing first 10 images from PREPROCESSED_DIR:\")\n#plt.figure(figsize=(15, 5))\n\n#for i, row in enumerate(aug_df.head(10).itertuples(), 1):\n#    img_path = os.path.join(PREPROCESSED_DIR, row.image)\n#    img = cv2.imread(img_path)      #img = cv2.imread(img_path, cv2.IMREAD_GRAYSCALE)\n    \n#    if img is not None:\n#        plt.subplot(2, 5, i)\n#        plt.imshow(img)      #plt.imshow(img, cmap='gray')\n#        plt.title(f\"Label: {row.level}\")\n#        plt.axis('off')\n\n#plt.tight_layout()\n#plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:40:25.321026Z","iopub.execute_input":"2025-04-30T20:40:25.321246Z","iopub.status.idle":"2025-04-30T20:40:25.324877Z","shell.execute_reply.started":"2025-04-30T20:40:25.321228Z","shell.execute_reply":"2025-04-30T20:40:25.324131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# STEP 3: Feature Extraction using ResNet50 (without classification head)\n# =====================\n\nfrom tensorflow.keras.applications import ResNet50, InceptionV3, EfficientNetB7, VGG16, VGG19, EfficientNetV2M\nfrom tensorflow.keras.applications.inception_v3 import preprocess_input as inception_preprocess\nfrom tensorflow.keras.applications.efficientnet import preprocess_input as efficientnet_preprocess\nfrom tensorflow.keras.applications.vgg16 import preprocess_input as vgg_preprocess\nfrom tensorflow.keras.applications.efficientnet_v2 import preprocess_input as efficientnetv2_preprocess\n\n\nfrom tensorflow.keras.applications.resnet50 import preprocess_input as resnet_preprocess\nfrom tensorflow.keras.preprocessing.image import img_to_array, load_img\nfrom sklearn.preprocessing import LabelEncoder\n\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, Dropout, Input\nfrom tensorflow.keras.utils import to_categorical\nfrom sklearn.preprocessing import LabelEncoder\n\ndef fine_tune_inceptionV3_head(img_size, aug_df, preprocessed_dir):\n    # Load and preprocess images\n    img_paths = [os.path.join(preprocessed_dir, fname) for fname in aug_df['image']]\n    labels = aug_df['level'].values\n    le = LabelEncoder()\n    labels_encoded = le.fit_transform(labels)\n    labels_cat = to_categorical(labels_encoded, num_classes=5)\n\n    X = []\n    for path in img_paths:\n        img = load_img(path, target_size=(img_size, img_size))\n        img = img_to_array(img)\n        img = inception_preprocess(img)\n        X.append(img)\n    X = np.array(X)\n\n    # Build model\n    base_input = Input(shape=(img_size, img_size, 3))\n    base_model = InceptionV3(include_top=False, weights='imagenet', pooling='avg', input_tensor=base_input)\n    x = base_model.output\n    x = Dropout(0.5)(x)\n    output = Dense(5, activation='softmax')(x)\n\n    model = Model(inputs=base_model.input, outputs=output)\n\n    # Train head\n    for layer in base_model.layers:\n        layer.trainable = False\n    model.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n    print(\"Training classification head...\")\n    model.fit(X, labels_cat, batch_size=32, epochs=10, validation_split=0.1, verbose=1)\n\n    # Fine-tune last layers\n    for layer in base_model.layers[-30:]:\n        layer.trainable = True\n    model.compile(optimizer=tf.keras.optimizers.Adam(1e-5), loss='categorical_crossentropy', metrics=['accuracy'])\n    print(\"Fine-tuning last 30 layers...\")\n    model.fit(X, labels_cat, batch_size=32, epochs=10, validation_split=0.1, verbose=1)\n\n    # Return the base model for feature extraction\n    return Model(inputs=base_model.input, outputs=base_model.output)\n\n\n\nFEATURES_DIR = 'features'\nos.makedirs(FEATURES_DIR, exist_ok=True)\n\nInceptionV3_model = fine_tune_inceptionV3_head(IMG_SIZE, aug_df, PREPROCESSED_DIR)\n\n\nfeature_rows = []\n\nfor idx, row in aug_df.iterrows():\n    img_path = os.path.join(PREPROCESSED_DIR, row['image'])\n    img = load_img(img_path, target_size=(IMG_SIZE, IMG_SIZE))\n    img = img_to_array(img)\n    img = np.expand_dims(img, axis=0)\n    img = inception_preprocess(img)\n    feature = InceptionV3_model.predict(img, verbose=0)[0]\n\n    feature_name = row['image'].replace('.png', '.npy')\n    np.save(os.path.join(FEATURES_DIR, feature_name), feature)\n\n    feature_rows.append({'feature': feature_name, 'level': row['level']})\n\nfeatures_df = pd.DataFrame(feature_rows)\nfeatures_df.to_csv('features.csv', index=False)\n\nprint(\"Feature extraction complete...\")\n\n#give seperate for each model what i have to change to make it work with inceptionv3, yolov8, efficientnetb7, vgg16, vgg19,efficientnetv2M. ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:40:25.325638Z","iopub.execute_input":"2025-04-30T20:40:25.326387Z","iopub.status.idle":"2025-04-30T20:49:18.241189Z","shell.execute_reply.started":"2025-04-30T20:40:25.326344Z","shell.execute_reply":"2025-04-30T20:49:18.240333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_list = []\n# Iterate through the feature_df\nfor idx, row in features_df.iterrows():\n    file_name = row['feature']\n    file_path = os.path.join(FEATURES_DIR, file_name)\n    data = np.load(file_path)\n    feature_list.append(data)\n\nfeature_list = np.array(feature_list)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:49:18.242074Z","iopub.execute_input":"2025-04-30T20:49:18.242342Z","iopub.status.idle":"2025-04-30T20:49:19.000008Z","shell.execute_reply.started":"2025-04-30T20:49:18.242311Z","shell.execute_reply":"2025-04-30T20:49:18.999200Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# STEP 4: Plot 2 Features from Each Class\n# =====================\n\nimport seaborn as sns\n\nfeatures_per_class = {i: [] for i in range(5)}\n\nfor idx, row in features_df.iterrows():\n    if len(features_per_class[row['level']]) < 2:\n        feat = np.load(os.path.join(FEATURES_DIR, row['feature']))\n        features_per_class[row['level']].append(feat)\n\nplt.figure(figsize=(20, 10))\n\nfor cls in range(5):\n    for i, feat in enumerate(features_per_class[cls]):\n        plt.subplot(5, 2, cls*2+i+1)\n        sns.histplot(feat, bins=30, kde=True)\n        plt.title(f'Class {cls} Feature {i+1}')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:49:19.000999Z","iopub.execute_input":"2025-04-30T20:49:19.001290Z","iopub.status.idle":"2025-04-30T20:49:22.364557Z","shell.execute_reply.started":"2025-04-30T20:49:19.001264Z","shell.execute_reply":"2025-04-30T20:49:22.363805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# First split: 80% train, 20% temp (val + test)\ntrain_dt, temp_dt, train_df, temp_df = train_test_split(feature_list, features_df, test_size=0.3, random_state=SEED, stratify=features_df['level'])\n\n# Second split: 50% of temp = 10% of total for val and test each\nval_dt, test_dt, val_df, test_df = train_test_split(temp_dt, temp_df, test_size=0.5, random_state=SEED, stratify=temp_df['level'])\n\n# Save CSVs\ntrain_df.to_csv('train_split.csv', index=False)\nval_df.to_csv('val_split.csv', index=False)\ntest_df.to_csv('test_split.csv', index=False)\n\n# Print summary\nprint(\"Train / Validation / Test splits saved as CSVs:\")\nprint(f\"Train: {len(train_df)} samples\")\nprint(f\"Validation: {len(val_df)} samples\")\nprint(f\"Test: {len(test_df)} samples\")\n\n# Class distribution in each split\nprint(\"\\nClass distribution in Train:\")\nprint(train_df['level'].value_counts().sort_index())\nprint(\"\\nClass distribution in Validation:\")\nprint(val_df['level'].value_counts().sort_index())\nprint(\"\\nClass distribution in Test:\")\nprint(test_df['level'].value_counts().sort_index())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:49:22.367394Z","iopub.execute_input":"2025-04-30T20:49:22.368171Z","iopub.status.idle":"2025-04-30T20:49:22.398765Z","shell.execute_reply.started":"2025-04-30T20:49:22.368141Z","shell.execute_reply":"2025-04-30T20:49:22.398044Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# STEP 5: Apply SMOTE for classes 1,2,3,4\n# =====================\n\nselected_classes = [ 1, 2, 3, 4]\nselected_features = []\nselected_labels = []\n\nfor idx, row in train_df.iterrows():\n    if row['level'] in selected_classes:\n        feat = np.load(os.path.join(FEATURES_DIR, row['feature']))\n        selected_features.append(feat)\n        selected_labels.append(row['level'])\n\nselected_features = np.array(selected_features)\nselected_labels = np.array(selected_labels)\n\n# Find the class with the highest count\nclass_counts = pd.Series(selected_labels).value_counts()\nhighest_count = class_counts.max()\n\nprint(\"Class counts before SMOTE:\")\nprint(class_counts)\n\n# Apply SMOTE\n#smote = SMOTE(random_state=SEED)\n#X_resampled, y_resampled = smote.fit_resample(selected_features, selected_labels)\n\n#adasyn\nfrom imblearn.over_sampling import ADASYN\nfrom imblearn.under_sampling import EditedNearestNeighbours\n\n# Step 1: Apply ADASYN\nadasyn = ADASYN(random_state=SEED, n_neighbors=5)\nX_adasyn, y_adasyn = adasyn.fit_resample(selected_features, selected_labels)      #X_resampled, y_resampled = adasyn.fit_resample(selected_features, selected_labels)                                        \n\n# Step 2: Apply ENN to clean noisy points\nenn = EditedNearestNeighbours()\nX_resampled, y_resampled = enn.fit_resample(X_adasyn, y_adasyn)\n#adasyn\n\n\n# Save synthetic features\nSYNTHETIC_DIR = 'synthetic_features'\nos.makedirs(SYNTHETIC_DIR, exist_ok=True)\n\nsynthetic_rows = []\n\ncount_original = len(selected_labels)\ncount_generated = len(X_resampled) - count_original\n\n# Saving only the synthetic samples\nsynthetic_start_idx = count_original\n\nfor i in range(synthetic_start_idx, len(X_resampled)):\n    feature = X_resampled[i]\n    label = y_resampled[i]\n    name = f'synthetic_{i}.npy'\n    np.save(os.path.join(SYNTHETIC_DIR, name), feature)\n    synthetic_rows.append({'feature': name, 'level': label})\n\nsynthetic_df = pd.DataFrame(synthetic_rows)\nsynthetic_df.to_csv('synthetic_features.csv', index=False)\n\nprint(\"Synthetic features generated for each class:\")\nprint(synthetic_df['level'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:57:11.809169Z","iopub.execute_input":"2025-04-30T20:57:11.809520Z","iopub.status.idle":"2025-04-30T20:57:13.354090Z","shell.execute_reply.started":"2025-04-30T20:57:11.809492Z","shell.execute_reply":"2025-04-30T20:57:13.353366Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# STEP 6: Plot 2 Synthetic Features from Each Class\n# =====================\n\nsynthetic_per_class = {i: [] for i in selected_classes}\n\nfor idx, row in synthetic_df.iterrows():\n    if len(synthetic_per_class[row['level']]) < 2:\n        feat = np.load(os.path.join(SYNTHETIC_DIR, row['feature']))\n        synthetic_per_class[row['level']].append(feat)\n\nplt.figure(figsize=(20, 10))\n\nfor cls in selected_classes:\n    for i, feat in enumerate(synthetic_per_class[cls]):\n        plt.subplot(4, 2, (cls-1)*2+i+1)\n        sns.histplot(feat, bins=30, kde=True)\n        plt.title(f'Synthetic Class {cls} Feature {i+1}')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:57:13.355292Z","iopub.execute_input":"2025-04-30T20:57:13.355641Z","iopub.status.idle":"2025-04-30T20:57:14.300749Z","shell.execute_reply.started":"2025-04-30T20:57:13.355622Z","shell.execute_reply":"2025-04-30T20:57:14.299966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# STEP 7: Prepare Final trainig Dataset\n# =====================\n\nfinal_features = []\nfinal_labels = []\n\n# Add original features\nfor idx, row in train_df.iterrows():\n    feat = np.load(os.path.join(FEATURES_DIR, row['feature']))\n    final_features.append(feat)\n    final_labels.append(row['level'])\n\n# Add synthetic features\nfor idx, row in synthetic_df.iterrows():\n    feat = np.load(os.path.join(SYNTHETIC_DIR, row['feature']))\n    final_features.append(feat)\n    final_labels.append(row['level'])\n\nfinal_features = np.array(final_features)\nfinal_labels = np.array(final_labels)\n\nprint(f\"Train: {len(final_features)}, Val: {len(val_dt)}, Test: {len(test_dt)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:57:14.301838Z","iopub.execute_input":"2025-04-30T20:57:14.302099Z","iopub.status.idle":"2025-04-30T20:57:14.908827Z","shell.execute_reply.started":"2025-04-30T20:57:14.302078Z","shell.execute_reply":"2025-04-30T20:57:14.908141Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# Final data Distribution \n# =====================\n\n# Count final class distribution\nfinal_class_counts = pd.Series(final_labels).value_counts().sort_index()\n\nprint(\"\\nFinal number of samples in each class:\")\nfor cls, count in final_class_counts.items():\n    print(f\"Class {cls}: {count} samples\")\n\n# Plot histogram of final class distribution\ncolors = ['#ff9999','#66b3ff','#99ff99','#ffcc99','#c2c2f0']  # Different colors for each class\n\nplt.figure(figsize=(8, 6))\nplt.bar(final_class_counts.index, final_class_counts.values, color=colors, edgecolor='black')\nplt.xticks(final_class_counts.index)\nplt.xlabel('Class Label')\nplt.ylabel('Number of Samples')\nplt.title('Final Class Distribution After Augmentation and SMOTE')\nplt.grid(axis='y', linestyle='--', alpha=0.7)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:57:14.909522Z","iopub.execute_input":"2025-04-30T20:57:14.909719Z","iopub.status.idle":"2025-04-30T20:57:15.056542Z","shell.execute_reply.started":"2025-04-30T20:57:14.909705Z","shell.execute_reply":"2025-04-30T20:57:15.055753Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 8: Train on sequential\n# =====================\n\ny_train = train_df['level'].values.astype(np.int32)\ny_val = val_df['level'].values.astype(np.int32)\n\n\n# Simple classifier on top of extracted features\nmodel = Sequential([\n    layers.Input(shape=(2048,)),\n    layers.Dense(256, activation='relu'),\n    layers.Dropout(0.5),\n    layers.Dense(5, activation='softmax')\n])\n\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=0.0001),\n    loss='SparseCategoricalCrossentropy',\n    metrics=['accuracy']\n)\n\n#model.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\ncallbacks = [\n    tf.keras.callbacks.ModelCheckpoint('best_model.keras', monitor='val_accuracy', save_best_only=True, mode='max'),\n    tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)\n]\n\nhistory = model.fit(\n    #train_dt, y_train,\n    final_features, final_labels,\n    validation_data=(val_dt, y_val),\n    epochs=100,\n    batch_size=64,\n    callbacks=callbacks,\n    verbose=1\n)\n\n\"\"\"# STEP 8: Train on sequential\n# =====================\n\ny_train = train_df['level'].values.astype(np.int32)\ny_val = val_df['level'].values.astype(np.int32)\n\n\n# Simple classifier on top of extracted features\nmodel = Sequential([\n    layers.Input(shape=(2048,)),\n    layers.Dense(512, activation='relu'),\n    layers.Dropout(0.3),\n    layers.Dense(5, activation='softmax')\n])\n\nmodel.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\ncallbacks = [\n    tf.keras.callbacks.ModelCheckpoint('best_model.keras', monitor='val_accuracy', save_best_only=True, mode='max'),\n    tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)\n]\n\nhistory = model.fit(\n    final_features, final_labels,\n    validation_data=(val_dt, y_val),\n    epochs=50,\n    batch_size=32,\n    callbacks=callbacks,\n    verbose=1\n)\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:57:42.761549Z","iopub.execute_input":"2025-04-30T20:57:42.761858Z","iopub.status.idle":"2025-04-30T20:57:50.913968Z","shell.execute_reply.started":"2025-04-30T20:57:42.761833Z","shell.execute_reply":"2025-04-30T20:57:50.913283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# STEP 9: Plot Accuracy and Loss\n# =====================\n\nplt.figure(figsize=(12, 5))\nplt.subplot(1,2,1)\nplt.plot(history.history['accuracy'], label='train_accuracy')\nplt.plot(history.history['val_accuracy'], label='val_accuracy')\nplt.legend()\nplt.title('Accuracy')\n\nplt.subplot(1,2,2)\nplt.plot(history.history['loss'], label='train_loss')\nplt.plot(history.history['val_loss'], label='val_loss')\nplt.legend()\nplt.title('Loss')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:58:16.641802Z","iopub.execute_input":"2025-04-30T20:58:16.642452Z","iopub.status.idle":"2025-04-30T20:58:16.985153Z","shell.execute_reply.started":"2025-04-30T20:58:16.642427Z","shell.execute_reply":"2025-04-30T20:58:16.984525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# STEP 10: Evaluate on Test Set\n# =====================\n\ny_pred = model.predict(test_dt)\ny_pred_labels = np.argmax(y_pred, axis=1)\n\nprint(classification_report(test_df['level'], y_pred_labels))\n\n# Confusion Matrix\ncm = confusion_matrix(test_df['level'], y_pred_labels)\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues')\nplt.title('Confusion Matrix')\nplt.xlabel('Predicted')\nplt.ylabel('True')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T21:16:54.136029Z","iopub.execute_input":"2025-04-30T21:16:54.136806Z","iopub.status.idle":"2025-04-30T21:16:54.429601Z","shell.execute_reply.started":"2025-04-30T21:16:54.136780Z","shell.execute_reply":"2025-04-30T21:16:54.428903Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score, roc_curve, auc\nfrom sklearn.preprocessing import label_binarize\n\n# Assuming `model` is your final trained model\n# and `test_df['level']` are the true labels\n\n# Predict probabilities for the test set\ny_true = test_df['level'].values\ny_score = model.predict(test_dt)\n\n# Binarize the output labels for multi-class AUC-ROC\ny_true_bin = label_binarize(y_true, classes=[0, 1, 2, 3, 4])\n\n# Compute ROC curve and ROC area for each class\nfpr = dict()\ntpr = dict()\nroc_auc = dict()\nfor i in range(5):\n    fpr[i], tpr[i], _ = roc_curve(y_true_bin[:, i], y_score[:, i])\n    roc_auc[i] = auc(fpr[i], tpr[i])\n\n# Plot all ROC curves\nplt.figure(figsize=(10, 8))\ncolors = ['#ff9999','#66b3ff','#99ff99','#ffcc99','#c2c2f0']\nfor i in range(5):\n    plt.plot(fpr[i], tpr[i], color=colors[i], lw=2,\n             label=f'Class {i} (AUC = {roc_auc[i]:0.2f})')\n\nplt.plot([0, 1], [0, 1], 'k--', lw=1)\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Multi-class ROC Curve')\nplt.legend(loc=\"lower right\")\nplt.grid(alpha=0.3)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T21:21:25.488708Z","iopub.execute_input":"2025-04-30T21:21:25.489014Z","iopub.status.idle":"2025-04-30T21:21:25.790794Z","shell.execute_reply.started":"2025-04-30T21:21:25.488993Z","shell.execute_reply":"2025-04-30T21:21:25.790083Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================================\n# STEP 4: GradCAM for 2 images per class\n# =========================================\n\nIMG_SIZE = 224  \nPREPROCESSED_DIR = 'processed_images'\n\nbase_model = ResNet50(weights='imagenet', include_top=False, pooling=None, input_shape=(IMG_SIZE, IMG_SIZE, 3))\nlast_conv_layer_name = 'conv5_block3_out'\n\ndef make_gradcam_heatmap(img_array, model, last_conv_layer_name):\n    grad_model = tf.keras.models.Model(\n        [model.inputs],\n        [model.get_layer(last_conv_layer_name).output, model.output]\n    )\n    with tf.GradientTape() as tape:\n        conv_outputs, predictions = grad_model(img_array)\n        loss = tf.reduce_mean(predictions)\n    grads = tape.gradient(loss, conv_outputs)\n    pooled_grads = tf.reduce_mean(grads, axis=(0, 1, 2))\n    conv_outputs = conv_outputs[0]\n    heatmap = conv_outputs @ pooled_grads[..., tf.newaxis]\n    heatmap = tf.squeeze(heatmap)\n    heatmap = tf.maximum(heatmap, 0) / tf.math.reduce_max(heatmap)\n    return heatmap.numpy()\n\ndef load_preprocessed_image(img_path):\n    img = tf.keras.preprocessing.image.load_img(img_path, target_size=(IMG_SIZE, IMG_SIZE))\n    img = tf.keras.preprocessing.image.img_to_array(img)\n    img = np.expand_dims(img, axis=0)\n    img = tf.keras.applications.resnet50.preprocess_input(img)\n    return img\n\n# Read your features dataframe (to know levels)\nfeatures_df = pd.read_csv('features.csv')\n\n# Pick 2 random images from each class\nselected_images = []\n\nfor level in sorted(features_df['level'].unique()):\n    level_df = features_df[features_df['level'] == level]\n    samples = level_df.sample(n=2, random_state=42)\n    selected_images.extend(samples['feature'].str.replace('.npy', '.png').tolist())\n\n# Apply GradCAM\nfor img_name in selected_images:\n    img_path = os.path.join(PREPROCESSED_DIR, img_name)\n    \n    # Preprocess\n    img_array = load_preprocessed_image(img_path)\n    \n    # Generate GradCAM heatmap\n    heatmap = make_gradcam_heatmap(img_array, base_model, last_conv_layer_name)\n    \n    # Resize heatmap\n    heatmap = cv2.resize(heatmap, (IMG_SIZE, IMG_SIZE))\n    heatmap = np.uint8(255 * heatmap)\n    heatmap = cv2.applyColorMap(heatmap, cv2.COLORMAP_JET)\n    \n    # Load original image\n    original_img = cv2.imread(img_path)\n    original_img = cv2.resize(original_img, (IMG_SIZE, IMG_SIZE))\n    \n    # Superimpose\n    superimposed_img = heatmap * 0.4 + original_img\n\n    # Plot\n    plt.figure(figsize=(8,4))\n    plt.suptitle(f\"GradCAM: {img_name}\", fontsize=14)\n    \n    plt.subplot(1,2,1)\n    plt.imshow(cv2.cvtColor(original_img, cv2.COLOR_BGR2RGB))\n    plt.title('Original')\n    plt.axis('off')\n    \n    plt.subplot(1,2,2)\n    plt.imshow(cv2.cvtColor(superimposed_img.astype('uint8'), cv2.COLOR_BGR2RGB))\n    plt.title('GradCAM Heatmap')\n    plt.axis('off')\n    \n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:57:24.966901Z","iopub.execute_input":"2025-04-30T20:57:24.967096Z","iopub.status.idle":"2025-04-30T20:57:31.831875Z","shell.execute_reply.started":"2025-04-30T20:57:24.967081Z","shell.execute_reply":"2025-04-30T20:57:31.831213Z"}},"outputs":[],"execution_count":null}]}