{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"},{"sourceId":2822650,"sourceType":"datasetVersion","datasetId":1715304}],"dockerImageVersionId":30919,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install tensorflow pennylane matplotlib pandas numpy scipy transformers imblearn scikit-image","metadata":{"_uuid":"f329e2f7-9a0f-4976-bb26-61e9e3d9b14d","_cell_guid":"e509845f-d997-4a9e-a027-f19490b91391","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-21T15:06:21.374375Z","iopub.execute_input":"2025-03-21T15:06:21.374848Z","iopub.status.idle":"2025-03-21T15:06:26.541947Z","shell.execute_reply.started":"2025-03-21T15:06:21.374810Z","shell.execute_reply":"2025-03-21T15:06:26.540713Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"a88e2a2b-6bcc-4703-882f-1c983d33ac25","_cell_guid":"aadd93d4-3791-4f96-98f3-2625ad1b32a1","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-21T15:06:26.543919Z","iopub.execute_input":"2025-03-21T15:06:26.544340Z","iopub.status.idle":"2025-03-21T15:06:26.549597Z","shell.execute_reply.started":"2025-03-21T15:06:26.544283Z","shell.execute_reply":"2025-03-21T15:06:26.548453Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport time\nimport random\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\nfrom scipy import ndimage\nfrom PIL import Image\nfrom collections import Counter\nfrom sklearn.metrics import accuracy_score, confusion_matrix, classification_report, roc_auc_score\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.utils import class_weight\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.feature_selection import SelectKBest, f_classif, mutual_info_classif, RFE\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.decomposition import PCA\nfrom imblearn.over_sampling import SMOTE\nfrom imblearn.combine import SMOTETomek\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import EfficientNetB4, ConvNeXtBase, ResNet50, DenseNet201, EfficientNetV2B3\nfrom tensorflow.keras.layers import Dense, Dropout, BatchNormalization, Input, GlobalAveragePooling2D, Concatenate, Multiply, Add\nfrom tensorflow.keras.models import Model, Sequential\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, EarlyStopping, ModelCheckpoint\nfrom tensorflow.keras.losses import SparseCategoricalCrossentropy\nfrom tensorflow.keras.metrics import AUC\nfrom transformers import ViTImageProcessor, TFViTForImageClassification\nfrom skimage import exposure\nimport joblib\nimport pennylane as qml\nfrom pennylane import numpy as pnp","metadata":{"_uuid":"371d782c-2ef7-4ff5-a9c5-2db6613d3d1d","_cell_guid":"097e07ef-6965-4284-b087-f683ce0465b2","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-21T15:06:26.552229Z","iopub.execute_input":"2025-03-21T15:06:26.552634Z","iopub.status.idle":"2025-03-21T15:06:26.577615Z","shell.execute_reply.started":"2025-03-21T15:06:26.552596Z","shell.execute_reply":"2025-03-21T15:06:26.576462Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def seed_everything(seed=42):\n    \"\"\"Set seeds for reproducibility\"\"\"\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n    os.environ['TF_DETERMINISTIC_OPS'] = '1'\n    os.environ['TF_CUDNN_DETERMINISTIC'] = '1'\n    \n    # Set NumPy print options for cleaner output\n    np.set_printoptions(precision=4, suppress=True)\n    \n    print(f\"Random seed set to {seed} ✓\")\n\nseed_everything()","metadata":{"_uuid":"3cd5de31-41a4-41c6-8067-a1b2aed8d950","_cell_guid":"8aa910b5-6219-4ece-b188-dcaf43b5a9fb","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-21T15:06:26.579692Z","iopub.execute_input":"2025-03-21T15:06:26.580070Z","iopub.status.idle":"2025-03-21T15:06:26.779605Z","shell.execute_reply.started":"2025-03-21T15:06:26.580028Z","shell.execute_reply":"2025-03-21T15:06:26.778525Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_SIZE = 224","metadata":{"_uuid":"2f34245a-ec11-47ef-87b3-77ae233dbfd2","_cell_guid":"130cfb8b-f72f-41be-9e66-6ff717aad599","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-21T15:06:26.781363Z","iopub.execute_input":"2025-03-21T15:06:26.781676Z","iopub.status.idle":"2025-03-21T15:06:26.804288Z","shell.execute_reply.started":"2025-03-21T15:06:26.781637Z","shell.execute_reply":"2025-03-21T15:06:26.803173Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_ROOT = '/kaggle/input/aptos2019'\nTRAIN_CSV = os.path.join(DATA_ROOT, 'train_1.csv')\nVAL_CSV = os.path.join(DATA_ROOT, 'valid.csv')\nTEST_CSV = os.path.join(DATA_ROOT, 'test.csv')\n\nTRAIN_IMG_DIR = os.path.join(DATA_ROOT, 'train_images/train_images')\nVAL_IMG_DIR = os.path.join(DATA_ROOT, 'val_images/val_images')\nTEST_IMG_DIR = os.path.join(DATA_ROOT, 'test_images/test_images')","metadata":{"_uuid":"1e580c99-e72a-42a6-85ab-f740fae7b744","_cell_guid":"85687211-a9c0-4629-b76b-ea99f11e0491","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-21T15:06:26.807281Z","iopub.execute_input":"2025-03-21T15:06:26.807733Z","iopub.status.idle":"2025-03-21T15:06:26.823916Z","shell.execute_reply.started":"2025-03-21T15:06:26.807702Z","shell.execute_reply":"2025-03-21T15:06:26.822601Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_qubits = 16 \ndev = qml.device(\"default.qubit\", wires=num_qubits)","metadata":{"_uuid":"e7aba1c6-310a-4ce3-9dd4-7cc569fd620d","_cell_guid":"bf98db42-5260-442e-81d2-a54083b60054","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-21T15:06:26.825236Z","iopub.execute_input":"2025-03-21T15:06:26.825677Z","iopub.status.idle":"2025-03-21T15:06:26.847963Z","shell.execute_reply.started":"2025-03-21T15:06:26.825629Z","shell.execute_reply":"2025-03-21T15:06:26.846711Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def enhanced_preprocess_image(path):\n    \"\"\"\n    Advanced preprocessing for retinal images with improved techniques:\n    1. Improved border detection and cropping\n    2. Multi-scale contrast enhancement\n    3. Adaptive circular masking with smooth transitions\n    4. Robust channel-wise normalization\n    \"\"\"\n    # Load image\n    img = Image.open(path).convert('RGB')\n    img_array = np.array(img)\n    \n    # More robust border detection and cropping\n    if np.mean(img_array) < 50:  # Likely has black borders\n        # Find non-black regions with more sensitivity\n        threshold = np.mean(img_array) + 10\n        y, x = np.where(np.mean(img_array, axis=2) > threshold)\n        if len(y) > 0 and len(x) > 0:\n            # Add padding to ensure no content is lost\n            padding = 10\n            y_min, y_max = max(0, min(y) - padding), min(img_array.shape[0], max(y) + padding)\n            x_min, x_max = max(0, min(x) - padding), min(img_array.shape[1], max(x) + padding)\n            if y_max > y_min and x_max > x_min:\n                img_array = img_array[y_min:y_max, x_min:x_max]\n                img = Image.fromarray(img_array)\n    \n    # Resize to target dimensions\n    img = img.resize((IMG_SIZE, IMG_SIZE))\n    img_array = np.array(img) / 255.0\n    \n    # Multi-scale contrast enhancement for better feature visibility\n    # 1. CLAHE for local contrast enhancement\n    img_adapteq = exposure.equalize_adapthist(img_array, clip_limit=0.03)\n    \n    # 2. Apply additional gamma correction for better feature visibility\n    img_adapteq = exposure.adjust_gamma(img_adapteq, 1.2)\n    \n    # 3. Apply local contrast normalization\n    for c in range(3):\n        channel = img_adapteq[:,:,c]\n        local_mean = ndimage.gaussian_filter(channel, sigma=15)\n        local_std = np.sqrt(ndimage.gaussian_filter(channel**2, sigma=15) - local_mean**2)\n        local_std[local_std < 0.001] = 1\n        img_adapteq[:,:,c] = (channel - local_mean) / local_std\n    \n    # Create refined circular mask (retina images are circular)\n    h, w = img_adapteq.shape[:2]\n    y, x = np.ogrid[:h, :w]\n    center = (h//2, w//2)\n    \n    # Adaptive radius based on image content\n    img_gray = np.mean(img_adapteq, axis=2)\n    non_zero = np.where(img_gray > 0.05)\n    if len(non_zero[0]) > 0:\n        # Calculate distance from center to farthest bright pixel\n        distances = np.sqrt((non_zero[0] - center[0])**2 + (non_zero[1] - center[1])**2)\n        radius = np.percentile(distances, 95)  # Use 95th percentile to avoid outliers\n    else:\n        radius = min(h, w) * 0.45\n    \n    mask = ((x - center[1])**2 + (y - center[0])**2) <= (radius**2)\n    \n    # Apply mask to each channel with smoother edge transition\n    masked_img = img_adapteq.copy()\n    for c in range(3):\n        channel = masked_img[:,:,c]\n        # Create smooth transition at the edge (soft masking)\n        edge_width = 5\n        soft_mask = np.ones_like(mask, dtype=float)\n        dist_from_center = np.sqrt((x - center[1])**2 + (y - center[0])**2)\n        transition_zone = (dist_from_center > radius - edge_width) & (dist_from_center < radius)\n        soft_mask[transition_zone] = 1 - (dist_from_center[transition_zone] - (radius - edge_width)) / edge_width\n        soft_mask[dist_from_center >= radius] = 0\n        \n        channel = channel * soft_mask\n        masked_img[:,:,c] = channel\n    \n    # Channel-wise standardization with robust statistics\n    for c in range(3):\n        channel = masked_img[:,:,c]\n        # Use only non-zero (masked) values for statistics\n        non_zero_vals = channel[channel > 0]\n        if len(non_zero_vals) > 0:\n            mean_val = np.mean(non_zero_vals)\n            std_val = np.std(non_zero_vals)\n            if std_val > 0:\n                channel = (channel - mean_val) / std_val\n                # Clip outliers\n                channel = np.clip(channel, -3, 3)\n                # Rescale to [0, 1]\n                channel = (channel + 3) / 6\n                masked_img[:,:,c] = channel\n    \n    # Final normalization\n    masked_img = np.clip(masked_img, 0, 1)\n    \n    return masked_img","metadata":{"_uuid":"b04a7c79-6b04-4522-82b5-c8d9f5abcd6f","_cell_guid":"ad87e06c-86cf-4fa7-a6b8-f6f174f4f005","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-21T15:06:26.851538Z","iopub.execute_input":"2025-03-21T15:06:26.851848Z","iopub.status.idle":"2025-03-21T15:06:26.871490Z","shell.execute_reply.started":"2025-03-21T15:06:26.851822Z","shell.execute_reply":"2025-03-21T15:06:26.870430Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_data(csv_path, image_dir, id_column='id_code', target_column='diagnosis', max_samples=None):\n    \"\"\"Load dataset from CSV and image directory\"\"\"\n    # Load CSV\n    df = pd.read_csv(csv_path)\n    \n    # Limit samples if needed\n    if max_samples:\n        df = df.head(max_samples)\n    \n    # Create full image paths\n    df['image_path'] = df[id_column].apply(lambda x: os.path.join(image_dir, f\"{x}.png\"))\n    \n    # Check that a few files exist\n    sample_paths = df['image_path'].head(3).tolist()\n    for path in sample_paths:\n        if not os.path.exists(path):\n            print(f\"Warning: File not found: {path}\")\n    \n    return df","metadata":{"_uuid":"fc0e5415-c2d8-4e9e-ba94-4fd1a1c47928","_cell_guid":"89f2aa5c-e4b2-49ee-ad8b-58eeb48a2e5b","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-21T15:06:26.873456Z","iopub.execute_input":"2025-03-21T15:06:26.873805Z","iopub.status.idle":"2025-03-21T15:06:26.895850Z","shell.execute_reply.started":"2025-03-21T15:06:26.873778Z","shell.execute_reply":"2025-03-21T15:06:26.894922Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def prepare_images(df, max_samples=None):\n    \"\"\"Prepare images for the model\"\"\"\n    if max_samples:\n        df = df.head(max_samples)\n    \n    # Load and preprocess images\n    images = []\n    print(f\"Processing {len(df)} images...\")\n    for i, path in enumerate(df['image_path']):\n        if i % 20 == 0:\n            print(f\"  Processed {i}/{len(df)} images\")\n        try:\n            img = enhanced_preprocess_image(path)\n            images.append(img)\n        except Exception as e:\n            print(f\"Error processing image {path}: {str(e)}\")\n            # Add a placeholder black image\n            images.append(np.zeros((IMG_SIZE, IMG_SIZE, 3)))\n    \n    return np.array(images)","metadata":{"_uuid":"dc30e9cb-b346-4ed3-b0f8-13a0b1c4d8ac","_cell_guid":"c0f85bb2-892d-49a0-adc6-1da9d7c4c7b3","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-21T15:06:26.897002Z","iopub.execute_input":"2025-03-21T15:06:26.897416Z","iopub.status.idle":"2025-03-21T15:06:26.915272Z","shell.execute_reply.started":"2025-03-21T15:06:26.897384Z","shell.execute_reply":"2025-03-21T15:06:26.914112Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def random_blur(img):\n    # Apply random blur with varying intensity\n    sigma = random.uniform(0, 1.0)\n    return ndimage.gaussian_filter(img, sigma=(sigma, sigma, 0))\n\nenhanced_augmentation = ImageDataGenerator(\n    rotation_range=30,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=[0.8, 1.3],  # Wider zoom range\n    horizontal_flip=True,\n    vertical_flip=True,\n    fill_mode='reflect',\n    brightness_range=[0.7, 1.3],  # More aggressive brightness\n    channel_shift_range=0.2,\n    preprocessing_function=random_blur\n)","metadata":{"_uuid":"53edef05-839c-43dd-b22b-caeab75eb3a7","_cell_guid":"c13cc341-ffc0-41d0-85eb-0eb7015e4691","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-21T15:06:26.916448Z","iopub.execute_input":"2025-03-21T15:06:26.916748Z","iopub.status.idle":"2025-03-21T15:06:26.938346Z","shell.execute_reply.started":"2025-03-21T15:06:26.916723Z","shell.execute_reply":"2025-03-21T15:06:26.937294Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def setup_optimized_feature_extraction():\n    \"\"\"Initialize only the best feature extraction models\"\"\"\n    print(\"Loading optimized feature extraction models...\")\n    \n    # Use only the top-performing models - ViT and EfficientNet\n    vit_processor = ViTImageProcessor.from_pretrained('google/vit-base-patch16-224')\n    vit_model = TFViTForImageClassification.from_pretrained('google/vit-base-patch16-224')\n    \n    efficientnet_model = EfficientNetV2B3(include_top=False, weights='imagenet', pooling='avg',\n                                      input_shape=(IMG_SIZE, IMG_SIZE, 3))\n    \n    return {\n        'vit_processor': vit_processor,\n        'vit_model': vit_model,\n        'efficientnet': efficientnet_model\n    }\n\n\ndef extract_features_from_cnn(model, images, batch_size=32):\n    \"\"\"Extract features from a CNN model\"\"\"\n    print(f\"Extracting features using {model.name}...\")\n    return model.predict(images, batch_size=batch_size, verbose=1)\n\ndef extract_vit_features_fixed(vit_processor, vit_model, images, batch_size=16):\n    \"\"\"Extract features from Vision Transformer with do_rescale=False for preprocessed images\"\"\"\n    print(\"Extracting ViT features...\")\n    features = []\n    \n    # Process in smaller batches to avoid memory issues\n    for i in range(0, len(images), batch_size):\n        batch = images[i:i+batch_size]\n        # Add do_rescale=False to avoid warnings\n        inputs = vit_processor(images=list(batch), return_tensors=\"tf\", do_rescale=False)\n        outputs = vit_model(**inputs)\n        features.append(outputs.logits.numpy())\n        \n        # Print progress\n        if (i+batch_size) % (batch_size*10) == 0 or i+batch_size >= len(images):\n            print(f\"  Processed {min(i+batch_size, len(images))}/{len(images)} images\")\n    \n    return np.vstack(features)\n    ","metadata":{"_uuid":"b23e9cba-aa34-4166-95eb-0c6e437b7d9a","_cell_guid":"388b4c98-f2d7-4d87-8909-61f63b61e863","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-21T15:06:26.939456Z","iopub.execute_input":"2025-03-21T15:06:26.939823Z","iopub.status.idle":"2025-03-21T15:06:26.953984Z","shell.execute_reply.started":"2025-03-21T15:06:26.939795Z","shell.execute_reply":"2025-03-21T15:06:26.953123Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"@qml.qnode(dev)\ndef enhanced_quantum_transform(features):\n    \"\"\"\n    Enhanced quantum circuit for feature transformation with:\n    1. Improved feature scaling\n    2. Multi-scale entanglement\n    3. Multiple basis measurements\n    \"\"\"\n    # Scale features to appropriate range with improved scaling\n    scaled_features = np.tanh(features[:num_qubits] / 5.0) * np.pi  # Better scaling for quantum gates\n    \n    # Encoding layer - More sophisticated data embedding\n    for i in range(num_qubits):\n        qml.RX(scaled_features[i % len(scaled_features)], wires=i)\n        qml.RY(scaled_features[(i + 1) % len(scaled_features)], wires=i)\n        qml.RZ(scaled_features[(i + 2) % len(scaled_features)], wires=i)\n    \n    # Multi-scale entanglement - Create correlations at different scales\n    for layer in range(3):  # Deeper circuit with 3 layers\n        # Full entanglement layer\n        for i in range(num_qubits):\n            qml.CNOT(wires=[i, (i+1) % num_qubits])\n        \n        # Skip connections for longer-range correlations\n        for i in range(num_qubits):\n            qml.CNOT(wires=[i, (i+3) % num_qubits])\n        \n        # Rotation layer with adaptive amplitudes\n        for i in range(num_qubits):\n            weight = 0.1 / (layer + 1)  # Decreasing weights for deeper layers\n            qml.RX(scaled_features[i % len(scaled_features)] * weight, wires=i)\n            qml.RY(scaled_features[(i+1) % len(scaled_features)] * weight, wires=i)\n            qml.RZ(scaled_features[(i+2) % len(scaled_features)] * weight, wires=i)\n    \n    # Multi-basis measurements for richer feature extraction\n    # Z-basis measurements for all qubits\n    z_measurements = [qml.expval(qml.PauliZ(i)) for i in range(num_qubits)]\n    # X-basis for even qubits\n    x_measurements = [qml.expval(qml.PauliX(i)) for i in range(0, num_qubits, 2)]\n    # Y-basis for odd qubits\n    y_measurements = [qml.expval(qml.PauliY(i)) for i in range(1, num_qubits, 2)]\n    \n    # Combine all measurements\n    return z_measurements + x_measurements + y_measurements","metadata":{"_uuid":"72b861ab-3a85-4397-9cd7-bf04bb04d024","_cell_guid":"6af77981-0bcf-422a-a48a-422a7e9eeb66","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-21T15:06:26.955220Z","iopub.execute_input":"2025-03-21T15:06:26.955573Z","iopub.status.idle":"2025-03-21T15:06:26.976408Z","shell.execute_reply.started":"2025-03-21T15:06:26.955536Z","shell.execute_reply":"2025-03-21T15:06:26.975283Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def apply_quantum_transformation(features, feature_indices=None, batch_size=50):\n    \"\"\"Apply quantum transformation to selected feature indices\"\"\"\n    print(\"Applying quantum transformation...\")\n    if feature_indices is None:\n        # If no indices provided, use first num_qubits features\n        feature_indices = list(range(min(num_qubits * 2, features.shape[1])))\n\n    transformed = []\n    total_samples = len(features)\n\n    # Process in batches\n    for i in range(0, total_samples, batch_size):\n        end_idx = min(i + batch_size, total_samples)\n        batch = features[i:end_idx]\n\n        batch_transformed = []\n        for f in batch:\n            # Extract selected features\n            selected_features = f[feature_indices]\n            # Apply quantum transformation\n            q_features = enhanced_quantum_transform(selected_features)\n            batch_transformed.append(q_features)\n\n        transformed.extend(batch_transformed)\n\n        # Print progress\n        if (i + batch_size) % 500 == 0 or end_idx == total_samples:\n            print(f\"  Processed {end_idx}/{total_samples} samples\")\n\n    return np.array(transformed)","metadata":{"_uuid":"f30f8347-2436-495a-ad08-c948fb0f62de","_cell_guid":"0020c712-82c4-4190-99d2-fa23f01562b3","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-21T15:06:26.977683Z","iopub.execute_input":"2025-03-21T15:06:26.978069Z","iopub.status.idle":"2025-03-21T15:06:26.998110Z","shell.execute_reply.started":"2025-03-21T15:06:26.978033Z","shell.execute_reply":"2025-03-21T15:06:26.996993Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef optimized_feature_selection(X_train, y_train, X_val, X_test):\n    \"\"\"More aggressive feature selection to reduce dimensionality\"\"\"\n    print(\"\\nPerforming optimized feature selection...\")\n    \n    # Standardize features first for better selection\n    scaler = StandardScaler()\n    X_train_scaled = scaler.fit_transform(X_train)\n    X_val_scaled = scaler.transform(X_val)\n    X_test_scaled = scaler.transform(X_test)\n    \n    # Apply PCA with variance retention\n    pca = PCA(n_components=0.95)  # Retain 95% of variance\n    X_train_pca = pca.fit_transform(X_train_scaled)\n    X_val_pca = pca.transform(X_val_scaled)\n    X_test_pca = pca.transform(X_test_scaled)\n    \n    print(f\"PCA reduced features from {X_train.shape[1]} to {X_train_pca.shape[1]} while retaining 95% variance\")\n    \n    # Select only top 300 features max (much more aggressive reduction)\n    n_features = min(300, X_train_pca.shape[1])\n    \n    # Use mutual information for feature selection\n    selector = SelectKBest(mutual_info_classif, k=n_features)\n    X_train_selected = selector.fit_transform(X_train_pca, y_train)\n    X_val_selected = selector.transform(X_val_pca)\n    X_test_selected = selector.transform(X_test_pca)\n    \n    print(f\"Final feature count: {X_train_selected.shape[1]}\")\n    \n    return X_train_selected, X_val_selected, X_test_selected, scaler, pca, selector","metadata":{"_uuid":"5ba1f04b-3b2d-4d2e-bbef-ddc16a41fa3f","_cell_guid":"36588135-74ce-4abb-9930-af6599616062","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-21T15:06:26.999276Z","iopub.execute_input":"2025-03-21T15:06:26.999656Z","iopub.status.idle":"2025-03-21T15:06:27.018645Z","shell.execute_reply.started":"2025-03-21T15:06:26.999625Z","shell.execute_reply":"2025-03-21T15:06:27.017572Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def perform_advanced_class_balancing(X, y):\n    \"\"\"\n    Enhanced class balancing strategy:\n    1. SMOTETomek for initial balancing\n    2. Targeted synthetic samples for minority classes\n    3. Computed class weights for training\n    \n    Handles small sample sizes by using only class weights when needed.\n    \"\"\"\n    print(\"\\nApplying advanced class balancing...\")\n    \n    # Convert to float32 to reduce memory usage\n    X = X.astype(np.float32)\n    \n    # Check for minimum samples per class required for SMOTE\n    unique_classes = np.unique(y)\n    class_counts = np.bincount(y)\n    min_samples = np.min(class_counts[class_counts > 0])\n    \n    # Simple class weighting for very small datasets\n    if min_samples < 5:\n        print(\"Dataset too small for SMOTE - using class weights only\")\n        # Calculate class weights\n        class_weights = class_weight.compute_class_weight(\n            class_weight='balanced',\n            classes=unique_classes,\n            y=y\n        )\n        class_weight_dict = {int(cls): float(weight) for cls, weight in zip(unique_classes, class_weights)}\n        \n        print(\"\\nClass distribution (original):\")\n        for cls in unique_classes:\n            class_name = {\n                0: 'No DR',\n                1: 'Mild DR',\n                2: 'Moderate DR',\n                3: 'Severe DR',\n                4: 'Proliferative DR'\n            }.get(cls, f'Class {cls}')\n            count = np.sum(y == cls)\n            print(f\"  Class {cls} ({class_name}): {count} samples\")\n        \n        print(\"\\nClass weights for training:\")\n        for cls, weight in class_weight_dict.items():\n            print(f\"  Class {cls}: {weight:.4f}\")\n        \n        return X, y, class_weight_dict\n    \n    # Check if we need to use a modified strategy for small but sufficient samples\n    if min_samples < 10:\n        print(\"Using modified resampling for small dataset\")\n        # Use SMOTE with reduced k_neighbors\n        smote = SMOTE(random_state=42, k_neighbors=min(min_samples-1, 3))\n        try:\n            X_resampled, y_resampled = smote.fit_resample(X, y)\n        except ValueError as e:\n            print(f\"SMOTE failed: {str(e)}\")\n            print(\"Falling back to simple class weights\")\n            # Calculate class weights\n            class_weights = class_weight.compute_class_weight(\n                class_weight='balanced',\n                classes=unique_classes,\n                y=y\n            )\n            class_weight_dict = {int(cls): float(weight) for cls, weight in zip(unique_classes, class_weights)}\n            return X, y, class_weight_dict\n    else:\n        # Apply SMOTETomek for better balance\n        try:\n            smote_tomek = SMOTETomek(random_state=42)\n            X_resampled, y_resampled = smote_tomek.fit_resample(X, y)\n        except ValueError as e:\n            print(f\"SMOTETomek failed: {str(e)}\")\n            # Fall back to just SMOTE\n            print(\"Falling back to plain SMOTE\")\n            smote = SMOTE(random_state=42)\n            try:\n                X_resampled, y_resampled = smote.fit_resample(X, y)\n            except ValueError as e:\n                print(f\"SMOTE failed too: {str(e)}\")\n                print(\"Using original data with class weights\")\n                # Calculate class weights\n                class_weights = class_weight.compute_class_weight(\n                    class_weight='balanced',\n                    classes=unique_classes,\n                    y=y\n                )\n                class_weight_dict = {int(cls): float(weight) for cls, weight in zip(unique_classes, class_weights)}\n                return X, y, class_weight_dict\n    \n    # Additional synthetic samples for minority classes\n    minority_classes = []\n    for cls in unique_classes:\n        if class_counts[cls] < np.median(class_counts[class_counts > 0]):\n            minority_classes.append(cls)\n    \n    if len(minority_classes) > 0 and X_resampled.shape[0] > 20:  # Only if we have enough samples\n        print(f\"Generating additional samples for minority classes: {minority_classes}\")\n        extra_samples = []\n        extra_labels = []\n        \n        for cls in minority_classes:\n            cls_indices = np.where(y_resampled == cls)[0]\n            if len(cls_indices) < 3:  # Skip if too few samples\n                continue\n                \n            X_cls = X_resampled[cls_indices]\n            \n            # Create additional synthetic samples with advanced noise\n            augmentation_factor = min(2, 20 // len(cls_indices))  # Limit augmentation for small datasets\n            for _ in range(augmentation_factor):\n                for i in range(len(X_cls)):\n                    # Add random noise and scaling\n                    noise_level = np.random.uniform(0.01, 0.05)\n                    scale_factor = np.random.uniform(0.95, 1.05)\n                    \n                    # Add correlated noise for better samples\n                    noise = np.random.normal(0, noise_level, size=X_cls[i].shape)\n                    # Add some correlation to the noise\n                    noise = ndimage.gaussian_filter(noise, sigma=1.0) * noise_level\n                    \n                    x_aug = X_cls[i] * scale_factor + noise\n                    extra_samples.append(x_aug)\n                    extra_labels.append(cls)\n        \n        # Add extra samples to dataset\n        if extra_samples:\n            X_resampled = np.vstack([X_resampled, np.array(extra_samples)])\n            y_resampled = np.concatenate([y_resampled, np.array(extra_labels)])\n    \n    # Check class distribution after resampling\n    unique_classes, class_counts = np.unique(y_resampled, return_counts=True)\n    class_distribution = {}\n    print(\"\\nClass distribution after resampling:\")\n    for cls, count in zip(unique_classes, class_counts):\n        class_name = {\n            0: 'No DR',\n            1: 'Mild DR',\n            2: 'Moderate DR',\n            3: 'Severe DR',\n            4: 'Proliferative DR'\n        }.get(cls, f'Class {cls}')\n        class_distribution[cls] = count\n        print(f\"  Class {cls} ({class_name}): {count} samples\")\n    \n    # Compute class weights for training\n    y_resampled_clean = y_resampled.astype(np.int32)\n    unique_classes_clean = np.unique(y_resampled_clean)\n    \n    # Calculate class weights using balanced approach\n    class_weights = class_weight.compute_class_weight(\n        class_weight='balanced',\n        classes=unique_classes_clean,\n        y=y_resampled_clean\n    )\n    class_weight_dict = {int(cls): float(weight) for cls, weight in zip(unique_classes_clean, class_weights)}\n    \n    print(\"\\nClass weights for training:\")\n    for cls, weight in class_weight_dict.items():\n        print(f\"  Class {cls}: {weight:.4f}\")\n    \n    return X_resampled, y_resampled, class_weight_dict","metadata":{"_uuid":"b65706a5-cb8c-4883-b2dd-40af392429d0","_cell_guid":"da155b00-da11-4cda-8baf-c0868f23f13a","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-21T15:06:27.019966Z","iopub.execute_input":"2025-03-21T15:06:27.020402Z","iopub.status.idle":"2025-03-21T15:06:27.044309Z","shell.execute_reply.started":"2025-03-21T15:06:27.020364Z","shell.execute_reply":"2025-03-21T15:06:27.043244Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def sparse_categorical_focal_loss(gamma=2.0, alpha=0.25):\n    \"\"\"\n    Focal loss for multi-class classification with integer labels\n    \"\"\"\n    def loss_function(y_true, y_pred):\n        # Cast y_true to int32\n        y_true = tf.cast(y_true, tf.int32)\n        \n        # Get the number of classes from y_pred shape\n        num_classes = tf.shape(y_pred)[-1]\n        \n        # Convert y_true to one-hot encoding\n        y_true_one_hot = tf.one_hot(y_true, depth=num_classes)\n        \n        # Clip predictions for numerical stability\n        epsilon = tf.keras.backend.epsilon()\n        y_pred = tf.clip_by_value(y_pred, epsilon, 1.0 - epsilon)\n        \n        # Calculate focal weights\n        # Higher gamma gives more weight to misclassified examples\n        focal_weight = tf.pow(1.0 - y_pred, gamma)\n        \n        # Apply class weights via alpha\n        if alpha is not None:\n            # For simplicity, use scalar alpha for all classes\n            # Could be extended to use different alpha per class\n            focal_weight = alpha * focal_weight * y_true_one_hot\n        \n        # Calculate cross entropy loss\n        ce_loss = -y_true_one_hot * tf.math.log(y_pred)\n        \n        # Apply focal weights to CE loss\n        focal_loss = focal_weight * ce_loss\n        \n        # Sum over classes, mean over batch\n        return tf.reduce_mean(tf.reduce_sum(focal_loss, axis=-1))\n    \n    return loss_function","metadata":{"_uuid":"d66ec16b-bf2c-4e40-859c-e0c602b615a0","_cell_guid":"c53b9d35-f9d8-45e2-b8d3-ddc1fe04c295","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-21T15:06:27.045201Z","iopub.execute_input":"2025-03-21T15:06:27.045562Z","iopub.status.idle":"2025-03-21T15:06:27.065969Z","shell.execute_reply.started":"2025-03-21T15:06:27.045525Z","shell.execute_reply":"2025-03-21T15:06:27.065026Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_optimized_classifier(input_shape, num_classes=5):\n    \"\"\"Simplified model with stronger regularization to combat overfitting\"\"\"\n    # Set input layer\n    input_layer = Input(shape=(input_shape,))\n    \n    # First dense block with strong regularization\n    x = Dense(256, activation='relu', \n              kernel_regularizer=tf.keras.regularizers.l2(0.01),\n              kernel_initializer='he_normal')(input_layer)\n    x = BatchNormalization()(x)\n    x = Dropout(0.5)(x)\n    \n    # Second dense block\n    x = Dense(128, activation='relu',\n              kernel_regularizer=tf.keras.regularizers.l2(0.01),\n              kernel_initializer='he_normal')(x)\n    x = BatchNormalization()(x)\n    x = Dropout(0.5)(x)\n    \n    # Output layer\n    output_layer = Dense(num_classes, activation='softmax',\n                        kernel_regularizer=tf.keras.regularizers.l2(0.01))(x)\n    \n    model = Model(inputs=input_layer, outputs=output_layer)\n    \n    # Use AdamW optimizer for better regularization\n    optimizer = tf.keras.optimizers.AdamW(\n        learning_rate=1e-3,\n        weight_decay=0.01\n    )\n    \n    model.compile(\n        optimizer=optimizer,\n        loss=tf.keras.losses.SparseCategoricalCrossentropy(),\n        metrics=['accuracy']\n    )\n    \n    return model","metadata":{"_uuid":"78b303ee-33ba-4d83-976a-56d705fd6a07","_cell_guid":"be51674a-60d8-44aa-b7ca-1737fc80bd7e","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-21T15:06:27.067219Z","iopub.execute_input":"2025-03-21T15:06:27.067636Z","iopub.status.idle":"2025-03-21T15:06:27.087611Z","shell.execute_reply.started":"2025-03-21T15:06:27.067598Z","shell.execute_reply":"2025-03-21T15:06:27.086442Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_with_optimized_strategy(model, X_train, y_train, X_val, y_val, class_weights=None):\n    \"\"\"Improved training strategy with aggressive early stopping and adaptive learning rates\"\"\"\n    \n    # Callbacks\n    # Early stopping based on validation loss with patience\n    early_stopping = EarlyStopping(\n        monitor='val_loss',\n        patience=10,\n        verbose=1,\n        restore_best_weights=True,\n        mode='min'\n    )\n    \n    # Reduce learning rate when stuck\n    reduce_lr = ReduceLROnPlateau(\n        monitor='val_loss',\n        factor=0.2,  # More aggressive reduction\n        patience=3,\n        min_lr=1e-6,\n        verbose=1,\n        mode='min'\n    )\n    \n    # Save best model\n    checkpoint = ModelCheckpoint(\n        'best_dr_model.keras',\n        monitor='val_loss',\n        save_best_only=True,\n        mode='min',\n        verbose=1\n    )\n    \n    # Compile with focal loss for better handling of class imbalance\n    model.compile(\n        optimizer=model.optimizer,\n        loss=sparse_categorical_focal_loss(gamma=2.0, alpha=0.25),\n        metrics=['accuracy']\n    )\n    \n    # Training\n    print(\"\\nTraining with optimized strategy...\")\n    history = model.fit(\n        X_train, y_train,\n        validation_data=(X_val, y_val),\n        epochs=100,  # Set high, early stopping will handle it\n        batch_size=32,\n        callbacks=[early_stopping, reduce_lr, checkpoint],\n        class_weight=class_weights,\n        verbose=1\n    )\n    \n    return model, history\n","metadata":{"_uuid":"1f3401df-c298-4a8f-884c-3914adfe6e53","_cell_guid":"d0d8d8e6-d93f-459d-9ce5-a4ff0f486ab6","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-21T15:06:27.088802Z","iopub.execute_input":"2025-03-21T15:06:27.089161Z","iopub.status.idle":"2025-03-21T15:06:27.111110Z","shell.execute_reply.started":"2025-03-21T15:06:27.089131Z","shell.execute_reply":"2025-03-21T15:06:27.110116Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_auc(model, X_test, y_test, num_classes=5):\n    \"\"\"Calculate AUC metrics manually with proper handling for sparse labels\"\"\"\n    # Get predictions\n    y_pred = model.predict(X_test)\n    \n    # Calculate AUC for each class (one-vs-rest)\n    auc_scores = []\n    for i in range(num_classes):\n        # Convert to binary classification problem\n        y_true_binary = (y_test == i).astype(int)\n        y_pred_binary = y_pred[:, i]\n        \n        # Check if we have both positive and negative examples\n        if np.sum(y_true_binary) > 0 and np.sum(y_true_binary) < len(y_true_binary):\n            try:\n                auc_i = roc_auc_score(y_true_binary, y_pred_binary)\n                auc_scores.append(auc_i)\n            except ValueError:\n                # Handle potential errors\n                print(f\"Could not calculate AUC for class {i}, using 0.5\")\n                auc_scores.append(0.5)\n        else:\n            # If only one class is present, AUC is undefined\n            print(f\"Class {i} has only one label value, AUC defaults to 0.5\")\n            auc_scores.append(0.5)\n    \n    # Calculate macro average\n    macro_auc = np.mean(auc_scores)\n    \n    print(\"\\nAUC Scores:\")\n    for i, score in enumerate(auc_scores):\n        print(f\"  Class {i}: {score:.4f}\")\n    print(f\"Macro AUC: {macro_auc:.4f}\")\n    \n    return {\n        'class_auc': auc_scores,\n        'macro_auc': macro_auc\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-21T15:06:27.112238Z","iopub.execute_input":"2025-03-21T15:06:27.112603Z","iopub.status.idle":"2025-03-21T15:06:27.133936Z","shell.execute_reply.started":"2025-03-21T15:06:27.112560Z","shell.execute_reply":"2025-03-21T15:06:27.132740Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_confusion_matrix(cm, class_names=None, title='Confusion Matrix'):\n    \"\"\"Plot confusion matrix with a heatmap\"\"\"\n    plt.figure(figsize=(10, 8))\n    import seaborn as sns\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues',\n                xticklabels=class_names, yticklabels=class_names)\n    plt.title(title)\n    plt.ylabel('True Label')\n    plt.xlabel('Predicted Label')\n    plt.tight_layout()\n    plt.show()\n\ndef plot_training_history(history):\n    \"\"\"Plot training history metrics\"\"\"\n    # Extract metrics\n    acc = history.history.get('accuracy', [])\n    val_acc = history.history.get('val_accuracy', [])\n    loss = history.history.get('loss', [])\n    val_loss = history.history.get('val_loss', [])\n    \n    # Plot accuracy and loss\n    fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(15, 5))\n    \n    # Accuracy plot\n    if acc:\n        ax1.plot(acc, label='Training Accuracy')\n    if val_acc:\n        ax1.plot(val_acc, label='Validation Accuracy')\n    ax1.set_title('Accuracy')\n    ax1.set_xlabel('Epoch')\n    ax1.set_ylabel('Accuracy')\n    ax1.legend()\n    \n    # Loss plot\n    if loss:\n        ax2.plot(loss, label='Training Loss')\n    if val_loss:\n        ax2.plot(val_loss, label='Validation Loss')\n    ax2.set_title('Loss')\n    ax2.set_xlabel('Epoch')\n    ax2.set_ylabel('Loss')\n    ax2.legend()\n    \n    plt.tight_layout()\n    plt.show()\n\ndef visualize_results(results):\n    \"\"\"Visualize model results\"\"\"\n    # Plot confusion matrix\n    cm = results['confusion_matrix']\n    class_names = ['No DR', 'Mild DR', 'Moderate DR', 'Severe DR', 'Proliferative DR']\n    plot_confusion_matrix(cm, class_names, 'DR Classification Confusion Matrix')\n    \n    # Plot training history\n    if 'history' in results:\n        plot_training_history(results['history'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-21T15:06:27.135514Z","iopub.execute_input":"2025-03-21T15:06:27.135964Z","iopub.status.idle":"2025-03-21T15:06:27.159382Z","shell.execute_reply.started":"2025-03-21T15:06:27.135919Z","shell.execute_reply":"2025-03-21T15:06:27.158380Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict_dr_severity(image_path, model, preprocessing_components):\n    \"\"\"Function to predict DR severity from a new image\"\"\"\n    # Preprocess image\n    img = enhanced_preprocess_image(image_path)\n    img = np.expand_dims(img, axis=0)\n    \n    # Extract features\n    # Load feature extraction models\n    models_dict = setup_optimized_feature_extraction()\n    vit_processor = models_dict['vit_processor']\n    vit_model = models_dict['vit_model']\n    \n    # Extract ViT features\n    vit_features = extract_vit_features_fixed(vit_processor, vit_model, img)\n    \n    # Extract EfficientNet features\n    efficient_features = extract_features_from_cnn(models_dict['efficientnet'], img)\n    \n    # Combine features\n    combined_features = np.hstack([vit_features, efficient_features])\n    \n    # Apply preprocessing transformations\n    scaler = preprocessing_components['scaler']\n    pca = preprocessing_components['pca']\n    selector = preprocessing_components['selector']\n    \n    # Transform features\n    scaled_features = scaler.transform(combined_features)\n    pca_features = pca.transform(scaled_features)\n    selected_features = selector.transform(pca_features)\n    \n    # Make prediction\n    predictions = model.predict(selected_features)\n    predicted_class = np.argmax(predictions, axis=1)[0]\n    \n    # Map to severity\n    severity_mapping = {\n        0: \"No DR\",\n        1: \"Mild DR\",\n        2: \"Moderate DR\",\n        3: \"Severe DR\",\n        4: \"Proliferative DR\"\n    }\n    \n    severity = severity_mapping[predicted_class]\n    confidence = predictions[0][predicted_class]\n    \n    return {\n        'severity': severity,\n        'severity_level': int(predicted_class),\n        'confidence': float(confidence),\n        'all_probabilities': predictions[0].tolist()\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-21T15:06:27.160534Z","iopub.execute_input":"2025-03-21T15:06:27.160908Z","iopub.status.idle":"2025-03-21T15:06:27.181519Z","shell.execute_reply.started":"2025-03-21T15:06:27.160874Z","shell.execute_reply":"2025-03-21T15:06:27.180437Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def run_optimized_pipeline(max_samples=None, use_quantum=False):\n    \"\"\"Optimized pipeline for diabetic retinopathy classification\"\"\"\n    # Start timing\n    import time\n    start_time = time.time()\n    \n    print(\"=\" * 50)\n    print(\"STARTING OPTIMIZED DR CLASSIFICATION PIPELINE\")\n    print(\"=\" * 50)\n    \n    # Step 1: Load data\n    print(\"\\n[1/6] Loading data...\")\n    train_df = pd.read_csv(TRAIN_CSV)\n    val_df = pd.read_csv(VAL_CSV)\n    test_df = pd.read_csv(TEST_CSV)\n    \n    # Limit samples if needed\n    if max_samples:\n        train_df = train_df.head(max_samples)\n        val_df = val_df.head(max(12, max_samples//8))\n        test_df = test_df.head(max(12, max_samples//8))\n    \n    # Create image paths\n    train_df['image_path'] = train_df['id_code'].apply(lambda x: f\"{TRAIN_IMG_DIR}/{x}.png\")\n    val_df['image_path'] = val_df['id_code'].apply(lambda x: f\"{VAL_IMG_DIR}/{x}.png\")\n    test_df['image_path'] = test_df['id_code'].apply(lambda x: f\"{TEST_IMG_DIR}/{x}.png\")\n    \n    print(f\"Dataset loaded: {len(train_df)} training, {len(val_df)} validation, {len(test_df)} test samples\")\n    \n    # Step 2: Process images with enhanced preprocessing\n    print(\"\\n[2/6] Preprocessing images...\")\n    train_images = []\n    val_images = []\n    test_images = []\n    \n    # Process training images\n    print(f\"Processing {len(train_df)} training images...\")\n    for i, path in enumerate(train_df['image_path']):\n        if i % 50 == 0 or i == len(train_df) - 1:\n            print(f\"  Processed {i+1}/{len(train_df)} images\")\n        try:\n            img = enhanced_preprocess_image(path)\n            train_images.append(img)\n        except Exception as e:\n            print(f\"Error processing image {path}: {str(e)}\")\n            train_images.append(np.zeros((IMG_SIZE, IMG_SIZE, 3)))\n    \n    # Process validation and test images\n    print(f\"Processing validation and test images...\")\n    for path in val_df['image_path']:\n        try:\n            img = enhanced_preprocess_image(path)\n            val_images.append(img)\n        except Exception as e:\n            print(f\"Error processing image {path}: {str(e)}\")\n            val_images.append(np.zeros((IMG_SIZE, IMG_SIZE, 3)))\n    \n    for path in test_df['image_path']:\n        try:\n            img = enhanced_preprocess_image(path)\n            test_images.append(img)\n        except Exception as e:\n            print(f\"Error processing image {path}: {str(e)}\")\n            test_images.append(np.zeros((IMG_SIZE, IMG_SIZE, 3)))\n    \n    train_images = np.array(train_images)\n    val_images = np.array(val_images)\n    test_images = np.array(test_images)\n    \n    # Step 3: Extract features using only the best models\n    print(\"\\n[3/6] Extracting features with optimized models...\")\n    models_dict = setup_optimized_feature_extraction()\n    \n    # Extract ViT features\n    vit_processor = models_dict['vit_processor']\n    vit_model = models_dict['vit_model']\n    \n    train_vit = extract_vit_features_fixed(vit_processor, vit_model, train_images)\n    val_vit = extract_vit_features_fixed(vit_processor, vit_model, val_images)\n    test_vit = extract_vit_features_fixed(vit_processor, vit_model, test_images)\n    \n    # Extract EfficientNet features\n    train_efficient = extract_features_from_cnn(models_dict['efficientnet'], train_images)\n    val_efficient = extract_features_from_cnn(models_dict['efficientnet'], val_images)\n    test_efficient = extract_features_from_cnn(models_dict['efficientnet'], test_images)\n    \n    # Combine features (only using the best two feature extractors)\n    train_combined = np.hstack([train_vit, train_efficient])\n    val_combined = np.hstack([val_vit, val_efficient])\n    test_combined = np.hstack([test_vit, test_efficient])\n    \n    print(f\"Combined features shape: {train_combined.shape}\")\n    \n    # Step 4: Enhanced feature selection (much more aggressive)\n    print(\"\\n[4/6] Performing optimized feature selection...\")\n    train_selected, val_selected, test_selected, scaler, pca, selector = optimized_feature_selection(\n        train_combined, train_df['diagnosis'].values, val_combined, test_combined\n    )\n    \n    # Step 5: Advanced class balancing\n    print(\"\\n[5/6] Performing advanced class balancing...\")\n    X_resampled, y_resampled, class_weight_dict = perform_advanced_class_balancing(\n        train_selected, train_df['diagnosis'].values\n    )\n    \n    # Step 6: Build and train optimized model\n    print(\"\\n[6/6] Training optimized model...\")\n    dr_classifier = build_optimized_classifier(X_resampled.shape[1], num_classes=5)\n    print(f\"Model built with {X_resampled.shape[1]} input features\")\n    \n    dr_classifier, history = train_with_optimized_strategy(\n        dr_classifier, \n        X_resampled, y_resampled,\n        val_selected, val_df['diagnosis'].values,\n        class_weight_dict\n    )\n    \n    # Evaluate model\n    print(\"\\nEvaluating optimized model on test set...\")\n    y_pred = dr_classifier.predict(test_selected)\n    y_pred_classes = np.argmax(y_pred, axis=1)\n    accuracy = accuracy_score(test_df['diagnosis'].values, y_pred_classes)\n    conf_matrix = confusion_matrix(test_df['diagnosis'].values, y_pred_classes)\n    \n    print(f\"\\nTest accuracy: {accuracy:.4f}\")\n    print(\"\\nConfusion matrix:\")\n    print(conf_matrix)\n    \n    # Calculate AUC manually\n    auc_results = calculate_auc(dr_classifier, test_selected, test_df['diagnosis'].values)\n    \n    # Save trained pipeline components\n    print(\"\\nSaving model and preprocessing components...\")\n    dr_classifier.save('dr_classifier_final.keras')\n    \n    # Save preprocessing components for inference\n    import pickle\n    with open('dr_preprocessing.pkl', 'wb') as f:\n        pickle.dump({\n            'scaler': scaler,\n            'pca': pca,\n            'selector': selector\n        }, f)\n    \n    # Calculate execution time\n    end_time = time.time()\n    total_time = end_time - start_time\n    hours, remainder = divmod(total_time, 3600)\n    minutes, seconds = divmod(remainder, 60)\n    \n    print(\"\\n\" + \"=\" * 50)\n    print(f\"OPTIMIZED PIPELINE COMPLETED SUCCESSFULLY!\")\n    print(f\"Total execution time: {int(hours)}h {int(minutes)}m {int(seconds)}s\")\n    print(\"=\" * 50)\n    \n    # Return results\n    return {\n        'model': dr_classifier,\n        'accuracy': accuracy,\n        'auc': auc_results['macro_auc'],\n        'auc_per_class': auc_results['class_auc'],\n        'confusion_matrix': conf_matrix,\n        'history': history,\n        'preprocessing': {\n            'scaler': scaler,\n            'pca': pca,\n            'selector': selector\n        }\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-21T15:06:27.184858Z","iopub.execute_input":"2025-03-21T15:06:27.185230Z","iopub.status.idle":"2025-03-21T15:06:27.206896Z","shell.execute_reply.started":"2025-03-21T15:06:27.185202Z","shell.execute_reply":"2025-03-21T15:06:27.205802Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results = run_optimized_pipeline(max_samples=None, use_quantum=False)\n\n# Visualize the results\nvisualize_results(results)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-21T15:06:27.207948Z","iopub.execute_input":"2025-03-21T15:06:27.208222Z","iopub.status.idle":"2025-03-21T16:04:36.160662Z","shell.execute_reply.started":"2025-03-21T15:06:27.208200Z","shell.execute_reply":"2025-03-21T16:04:36.158806Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}