{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"},{"sourceId":229480057,"sourceType":"kernelVersion"},{"sourceId":236869755,"sourceType":"kernelVersion"},{"sourceId":237896193,"sourceType":"kernelVersion"}],"dockerImageVersionId":31011,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport torch\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom torchvision import transforms\nfrom torch.utils.data import DataLoader, Dataset\nfrom PIL import Image, ImageFile\nfrom sklearn.metrics import classification_report, confusion_matrix, cohen_kappa_score, accuracy_score # Added accuracy_score\nimport torch.nn as nn\nimport torchvision.models as models\nfrom tqdm.notebook import tqdm # Use tqdm notebook version for better display in notebooks","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T11:02:45.606922Z","iopub.execute_input":"2025-04-29T11:02:45.607489Z","iopub.status.idle":"2025-04-29T11:02:57.877371Z","shell.execute_reply.started":"2025-04-29T11:02:45.607464Z","shell.execute_reply":"2025-04-29T11:02:57.876593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 1. Setup and Seed ---\nprint(\"--- Initializing ---\")\nseed = 42\ntorch.manual_seed(seed)\nif torch.cuda.is_available():\n    torch.cuda.manual_seed_all(seed)\nnp.random.seed(seed)\n\n# Optional: Deterministic behavior (can slow down training/inference)\n# torch.backends.cudnn.deterministic = True\n# torch.backends.cudnn.benchmark = False\n\n# Allow loading of truncated images\nImageFile.LOAD_TRUNCATED_IMAGES = True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T11:02:57.878572Z","iopub.execute_input":"2025-04-29T11:02:57.879282Z","iopub.status.idle":"2025-04-29T11:02:57.986017Z","shell.execute_reply.started":"2025-04-29T11:02:57.879262Z","shell.execute_reply":"2025-04-29T11:02:57.985251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 2. Configuration ---\nprint(\"--- Configuring Paths and Parameters ---\")\nAPTOS_CSV_PATH = '/kaggle/input/aptos2019-blindness-detection/test.csv'\nAPTOS_IMG_DIR = '/kaggle/input/aptos2019-blindness-detection/test_images/'\n\n# === PATH TO PRE-TRAINED BINARY MODEL (OUTPUTS 0 OR 1) ===\nBINARY_MODEL_PATH = '/kaggle/input/eyepacs-ddr-resnet-model/EyePacs_DDR_best_resnet_model.pth'\n\n# === PATH TO PRE-TRAINED MULTI-CLASS MODEL (OUTPUTS 0-3 for classes 1-4) ===\nMULTI_CLASS_MODEL_PATH = '/kaggle/input/multi-class-reg-model/Reg_QWK_Stop_resnet_model.pth'\n\nBATCH_SIZE_INFERENCE = 128 # Adjust based on GPU memory if needed\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nNUM_WORKERS = 2 # Number of workers for DataLoader\n\n# --- Determine Binary Model Output Size ---\nBINARY_NUM_CLASSES = 1 \nMULTI_NUM_CLASSES = 4 # Outputs 0, 1, 2, 3 (for original labels 1, 2, 3, 4)\n\nprint(f\"Using device: {DEVICE}\")\nprint(f\"APTOS CSV path: {APTOS_CSV_PATH}\")\nprint(f\"APTOS Image dir: {APTOS_IMG_DIR}\")\nprint(f\"Binary model path: {BINARY_MODEL_PATH}\")\nprint(f\"Multi-class model path: {MULTI_CLASS_MODEL_PATH}\")\nprint(f\"Binary model configured for {BINARY_NUM_CLASSES} output classes.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T11:02:57.986916Z","iopub.execute_input":"2025-04-29T11:02:57.987170Z","iopub.status.idle":"2025-04-29T11:02:57.992951Z","shell.execute_reply.started":"2025-04-29T11:02:57.987145Z","shell.execute_reply":"2025-04-29T11:02:57.992371Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 3. Transforms ---\nprint(\"--- Defining Image Transforms ---\")\ninference_transforms = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\nprint(\"Inference transforms defined.\")\n\n# --- 4. Model Definition ---\nprint(\"--- Defining Model Architecture ---\")\nclass ResNetModel(nn.Module):\n    def __init__(self, model_path=None, num_classes=1, load_weights=True, pretrained_backbone=False):\n        super(ResNetModel, self).__init__()\n        self.resnet = models.resnet18(pretrained=pretrained_backbone)  # Load ResNet18 backbone\n\n        # Remove the original FC layer\n        num_ftrs = self.resnet.fc.in_features\n        self.resnet.fc = nn.Identity()  # Remove the last layer\n\n        self.new_classifier = nn.Sequential( \n            nn.Linear(num_ftrs, 256),\n            nn.ReLU(),\n            nn.Dropout(0.5),\n            nn.Linear(256, 128),\n            nn.ReLU(),\n            nn.Dropout(0.5),\n            nn.Linear(128, num_classes) # Output layer size depends on model type\n        )\n        \n        if model_path and load_weights:\n            if not os.path.exists(model_path):\n                 print(f\"ERROR: Weight file not found at {model_path}. Model will be uninitialized.\")\n                 return\n            try:\n                print(f\"Loading weights from: {model_path} for {num_classes} classes\")\n                state_dict = torch.load(model_path, map_location=DEVICE)\n\n                load_result = self.load_state_dict(state_dict, strict=False) \n                print(f\"Weight loading result: {load_result}\")\n                if load_result.missing_keys:\n                    print(f\"Warning: Missing keys during load: {load_result.missing_keys}\")\n                if load_result.unexpected_keys:\n                    print(f\"Warning: Unexpected keys during load: {load_result.unexpected_keys}\")\n\n            except Exception as e:\n                print(f\"Error loading weights for {num_classes}-class model from {model_path}: {e}\")\n                print(\"Model will proceed with potentially uninitialized/partially loaded weights.\")\n\n    def forward(self, x):\n        x = self.resnet(x)\n        x = self.new_classifier(x)\n        return x\n\nprint(\"ResNetModel class defined.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T11:02:57.994552Z","iopub.execute_input":"2025-04-29T11:02:57.995152Z","iopub.status.idle":"2025-04-29T11:02:58.021629Z","shell.execute_reply.started":"2025-04-29T11:02:57.995133Z","shell.execute_reply":"2025-04-29T11:02:58.021079Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 5. Load Pre-trained Models ---\nprint(\"--- Loading Pre-trained Models ---\")\n# --- Load Binary Model ---\nprint(\"Loading Binary Model...\")\nbinary_model = ResNetModel(model_path=BINARY_MODEL_PATH, num_classes=BINARY_NUM_CLASSES)\nbinary_model.to(DEVICE)\nbinary_model.eval()\nprint(\"Binary model loaded and set to eval mode.\")\n\n# --- Load Multi-Class Model ---\nprint(\"\\nLoading Multi-Class Model...\")\nmulti_class_model = ResNetModel(model_path=MULTI_CLASS_MODEL_PATH, num_classes=MULTI_NUM_CLASSES)\nmulti_class_model.to(DEVICE)\nmulti_class_model.eval()\nprint(\"Multi-class model loaded and set to eval mode.\")\n\n# --- 6. APTOS Dataset and DataLoader ---\nprint(\"--- Preparing APTOS Data ---\")\nclass AptosDataset(Dataset):\n    def __init__(self, df, img_dir, transform=None, file_ext=\".png\"):\n        self.df = df\n        self.img_dir = img_dir\n        self.transform = transform\n        self.file_ext = file_ext\n        self.df.columns = self.df.columns.str.strip()\n        print(f\"Dataset initialized. Found columns: {self.df.columns.tolist()}\")\n        # --- MODIFICATION 1: Check only for id_code ---\n        if 'id_code' not in self.df.columns:\n            raise ValueError(\"CSV file must contain 'id_code' column.\")\n        # --- End MODIFICATION 1 ---\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        if idx >= len(self.df):\n             raise IndexError(f\"Index {idx} out of bounds for dataset of length {len(self.df)}\")\n        row = self.df.iloc[idx]\n        img_id = row[\"id_code\"]\n\n        # --- MODIFICATION 2: Handle missing 'diagnosis' column ---\n        # Assign a dummy label if 'diagnosis' is not present (e.g., in test.csv)\n        if 'diagnosis' in row and pd.notna(row['diagnosis']):\n            original_label = int(row[\"diagnosis\"]) # Use real label if available\n        else:\n            original_label = 0 # Assign dummy label 0 for test set or missing values\n        # --- End MODIFICATION 2 ---\n\n        img_filename = f\"{img_id}{self.file_ext}\"\n        img_path = os.path.join(self.img_dir, img_filename)\n        try:\n            image = Image.open(img_path).convert(\"RGB\")\n        except (OSError, IOError, FileNotFoundError) as e:\n            print(f\"Warning: Error loading image {img_path}: {e}. Returning None.\")\n            # Return None for image, but keep other info for collate_fn and potential tracking\n            return None, original_label, img_id, idx\n        if self.transform:\n            image = self.transform(image)\n        # Return image, the determined label (real or dummy), image_id, and original index\n        return image, original_label, img_id, idx\n\n# Load APTOS data (Ensure APTOS_CSV_PATH points to test.csv)\nif not os.path.exists(APTOS_CSV_PATH):\n    raise FileNotFoundError(f\"APTOS CSV file not found at {APTOS_CSV_PATH}\")\nif not os.path.isdir(APTOS_IMG_DIR):\n     raise FileNotFoundError(f\"APTOS Image directory not found at {APTOS_IMG_DIR}\")\n\naptos_df = pd.read_csv(APTOS_CSV_PATH)\nprint(f\"Loaded APTOS dataframe with {len(aptos_df)} samples from {APTOS_CSV_PATH}.\")\n\n# Create dataset\naptos_dataset = AptosDataset(df=aptos_df, img_dir=APTOS_IMG_DIR, transform=inference_transforms)\n\n# Define a collate function to handle None returns from dataset\ndef collate_fn_skip_none(batch):\n    # Filter out items where the image (item[0]) is None\n    filtered_batch = [item for item in batch if item[0] is not None]\n    if not filtered_batch:\n        return None # Return None if the entire batch failed to load\n    # Use default collate function on the filtered batch\n    return torch.utils.data.dataloader.default_collate(filtered_batch)\n\n# Create DataLoader\naptos_loader = DataLoader(\n    aptos_dataset,\n    batch_size=BATCH_SIZE_INFERENCE,\n    shuffle=False, # Keep shuffle=False for test set prediction order\n    num_workers=NUM_WORKERS,\n    collate_fn=collate_fn_skip_none, # Use the custom collate function\n    pin_memory=True if DEVICE.type == 'cuda' else False\n)\nprint(f\"APTOS DataLoader ready with {len(aptos_loader)} batches.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T11:02:58.022464Z","iopub.execute_input":"2025-04-29T11:02:58.023026Z","iopub.status.idle":"2025-04-29T11:03:00.364188Z","shell.execute_reply.started":"2025-04-29T11:02:58.023008Z","shell.execute_reply":"2025-04-29T11:03:00.363426Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 7. Stage 1: Binary Classification ---\nprint(\"\\n--- Starting Stage 1: Binary Classification ---\")\nstage1_no_dr_data = [] # List to store {'id_code': ..., 'diagnosis': 0}\nstage1_dr_ids = []     # List to store id_codes predicted as DR (1)\nimage_ids_processed = [] # Keep track of all processed image ids\noriginal_indices_processed = [] # Keep track of original indices if needed later\nskipped_batches_stage1 = 0\n\nwith torch.no_grad():\n    for batch_data in tqdm(aptos_loader, desc=\"Stage 1: Binary Prediction\"):\n        if batch_data is None:\n            print(\"Skipping a batch due to image loading errors in Stage 1.\")\n            skipped_batches_stage1 += 1\n            continue\n\n        # Unpack batch data - we don't need true_labels for test set prediction\n        inputs, _, img_ids_batch, original_indices_batch = batch_data\n        inputs = inputs.to(DEVICE)\n\n        outputs = binary_model(inputs)\n\n        # --- Determine Binary Prediction (0 or 1) ---\n        if BINARY_NUM_CLASSES == 2:\n            probs = torch.softmax(outputs, dim=1)\n            predicted_binary = torch.argmax(probs, dim=1) # 0 or 1\n        elif BINARY_NUM_CLASSES == 1:\n            probs = torch.sigmoid(outputs).squeeze(-1) # Ensure squeeze removes last dim if size 1\n            # Handle cases where batch size is 1 after filtering\n            if probs.ndim == 0:\n                 probs = probs.unsqueeze(0)\n            predicted_binary = (probs > 0.5).long() # 0 or 1\n        else:\n            raise ValueError(f\"Unsupported BINARY_NUM_CLASSES: {BINARY_NUM_CLASSES}.\")\n        # --------------------------------------------\n\n        predicted_binary_np = predicted_binary.cpu().numpy()\n        img_ids_list_batch = list(img_ids_batch)\n        original_indices_list_batch = original_indices_batch.cpu().numpy()\n\n        # --- Populate lists based on binary prediction ---\n        for i in range(len(predicted_binary_np)):\n            img_id = img_ids_list_batch[i]\n            prediction = predicted_binary_np[i]\n            original_idx = original_indices_list_batch[i]\n\n            image_ids_processed.append(img_id)\n            original_indices_processed.append(original_idx)\n\n            if prediction == 0:\n                stage1_no_dr_data.append({'id_code': img_id, 'diagnosis': 0})\n            else: # prediction == 1\n                stage1_dr_ids.append(img_id)\n        # -------------------------------------------------\n\nif skipped_batches_stage1 > 0:\n    print(f\"\\nWarning: Skipped {skipped_batches_stage1} batches in Stage 1 due to loading errors.\")\n    print(f\"Number of images processed ({len(image_ids_processed)}) might not match original dataset size ({len(aptos_df)}).\")\n\n# --- Create DataFrame for No DR predictions and Set for DR ids ---\ndf_no_dr = pd.DataFrame(stage1_no_dr_data)\ndr_image_ids_set = set(stage1_dr_ids) # Use a set for fast lookups in Stage 2\n\nprint(f\"Binary prediction finished.\")\nprint(f\"Found {len(df_no_dr)} No DR predictions (label 0).\")\nprint(f\"Found {len(dr_image_ids_set)} DR predictions (label 1) to be processed in Stage 2.\")\nprint(f\"Total images processed in Stage 1: {len(image_ids_processed)}\")\nif len(df_no_dr) + len(dr_image_ids_set) != len(image_ids_processed):\n     print(f\"Warning: Discrepancy in counts! NoDR ({len(df_no_dr)}) + DR ({len(dr_image_ids_set)}) != Processed ({len(image_ids_processed)})\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T11:03:00.365009Z","iopub.execute_input":"2025-04-29T11:03:00.365290Z","iopub.status.idle":"2025-04-29T11:04:05.182082Z","shell.execute_reply.started":"2025-04-29T11:03:00.365265Z","shell.execute_reply":"2025-04-29T11:04:05.181115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_no_dr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T11:04:05.183206Z","iopub.execute_input":"2025-04-29T11:04:05.183458Z","iopub.status.idle":"2025-04-29T11:04:05.212134Z","shell.execute_reply.started":"2025-04-29T11:04:05.183434Z","shell.execute_reply":"2025-04-29T11:04:05.211339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# dr_image_ids_set","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T11:04:05.212997Z","iopub.execute_input":"2025-04-29T11:04:05.213298Z","iopub.status.idle":"2025-04-29T11:04:05.224767Z","shell.execute_reply.started":"2025-04-29T11:04:05.213272Z","shell.execute_reply":"2025-04-29T11:04:05.223978Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 8. Stage 2: Multi-Class Classification (on DR Subset) ---\nprint(\"\\n--- Starting Stage 2: Multi-Class Classification ---\")\nstage2_results_data = [] # List to store {'id_code': ..., 'diagnosis': 1-4}\nprocessed_ids_stage2 = set()\nskipped_batches_stage2 = 0\n\nwith torch.no_grad():\n    for batch_data in tqdm(aptos_loader, desc=\"Stage 2: Multi-Class Prediction\"):\n        if batch_data is None:\n            print(\"Skipping a batch due to image loading errors in Stage 2.\")\n            skipped_batches_stage2 += 1\n            continue\n\n        # Unpack batch data\n        inputs, _, img_ids_batch, _ = batch_data # Don't need labels or original indices here\n\n        # --- Filter batch for images predicted as DR in Stage 1 ---\n        indices_to_process_in_batch = []\n        ids_to_process_in_batch = []\n        inputs_to_process_list = []\n\n        for i, img_id in enumerate(img_ids_batch):\n            if img_id in dr_image_ids_set:\n                indices_to_process_in_batch.append(i)\n                ids_to_process_in_batch.append(img_id)\n                inputs_to_process_list.append(inputs[i])\n        # ---------------------------------------------------------\n\n        # --- Proceed only if there are DR images in this batch ---\n        if not inputs_to_process_list:\n            continue # Skip to next batch if no DR images here\n\n        inputs_batch_multiclass = torch.stack(inputs_to_process_list).to(DEVICE)\n        # ---------------------------------------------------------\n\n        # --- Get Multi-Class Predictions ---\n        outputs_multi = multi_class_model(inputs_batch_multiclass)\n        _, predicted_multi_indices = torch.max(outputs_multi, 1) # Indices 0-3\n        # -----------------------------------\n\n        # --- Remap predictions (0-3) to DR scale (1-4) ---\n        predicted_multi_labels = predicted_multi_indices.cpu().numpy() + 1\n        # -------------------------------------------------\n\n        # --- Store results ---\n        for i in range(len(ids_to_process_in_batch)):\n            img_id = ids_to_process_in_batch[i]\n            multi_label = predicted_multi_labels[i]\n            stage2_results_data.append({'id_code': img_id, 'diagnosis': multi_label})\n            processed_ids_stage2.add(img_id)\n        # ---------------------\n\nif skipped_batches_stage2 > 0:\n    print(f\"\\nWarning: Skipped {skipped_batches_stage2} batches in Stage 2 due to loading errors.\")\n\n# --- Create DataFrame for Stage 2 results ---\ndf_stage2_results = pd.DataFrame(stage2_results_data)\n\nprint(f\"Multi-class prediction finished.\")\nprint(f\"Processed {len(processed_ids_stage2)} unique images in Stage 2.\")\nprint(f\"Generated {len(df_stage2_results)} multi-class predictions (labels 1-4).\")\n\n# --- Verification ---\nif len(processed_ids_stage2) != len(dr_image_ids_set):\n    print(f\"Warning: Mismatch! Number of unique IDs processed in Stage 2 ({len(processed_ids_stage2)}) \"\n          f\"does not match the number of DR IDs from Stage 1 ({len(dr_image_ids_set)}).\")\n    missed_in_stage2 = dr_image_ids_set - processed_ids_stage2\n    if missed_in_stage2:\n         print(f\"-> {len(missed_in_stage2)} DR IDs from Stage 1 were not processed in Stage 2 (likely due to skipped batches). Example missed IDs: {list(missed_in_stage2)[:10]}\")\nelse:\n    print(\"Verification successful: All DR IDs from Stage 1 were processed in Stage 2.\")\n\nif not df_stage2_results.empty:\n    print(\"\\nSample of Stage 2 Results (Multi-class):\")\n    print(df_stage2_results.head())\nelse:\n    print(\"\\nNo Stage 2 results generated (either no DR images found or all were in skipped batches).\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T11:04:05.225664Z","iopub.execute_input":"2025-04-29T11:04:05.225926Z","iopub.status.idle":"2025-04-29T11:04:48.906225Z","shell.execute_reply.started":"2025-04-29T11:04:05.225900Z","shell.execute_reply":"2025-04-29T11:04:48.904998Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 9. Combine Results & Visualize Prediction Distribution ---\nprint(\"\\n--- Combining Stage 1 and Stage 2 Results ---\")\n\n# Concatenate the DataFrames\nif not df_stage2_results.empty:\n    final_results_df = pd.concat([df_no_dr, df_stage2_results], ignore_index=True)\nelse:\n    print(\"Warning: Stage 2 results DataFrame is empty. Using only Stage 1 No DR results.\")\n    final_results_df = df_no_dr.copy()\n\n\nprint(f\"Combined DataFrame created with {len(final_results_df)} entries.\")\nprint(\"Sample of combined results:\")\nprint(final_results_df.head())\nprint(\"\\nValue counts in combined results:\")\nprint(final_results_df['diagnosis'].value_counts().sort_index())\n\n# --- Distribution of Final Predictions (Still Useful) ---\n# This plot shows the distribution of the model's predictions on the processed test set samples.\ntarget_names_aptos = ['No DR (0)', 'Mild DR (1)', 'Moderate DR (2)', 'Severe DR (3)', 'Proliferative DR (4)']\n\nplt.figure(figsize=(8, 6))\n# Use the 'diagnosis' column from the combined DataFrame\nsns.countplot(x='diagnosis', data=final_results_df, order=np.arange(len(target_names_aptos)), palette=\"viridis\")\nplt.xticks(ticks=np.arange(len(target_names_aptos)), labels=target_names_aptos, rotation=45, ha='right')\nplt.xlabel(\"Predicted Class (Combined Stages)\")\nplt.ylabel(\"Count\")\nplt.title(\"Distribution of Final Combined Predictions on Processed Test Set\")\nplt.tight_layout()\nplt.grid(axis=\"y\", linestyle='--', alpha=0.7)\nplt.savefig(\"final_prediction_distribution_test.png\")\nplt.show()\n\nprint(\"\\nFinished Combining and Visualization Section.\")\n# Note: Evaluation metrics like QWK, classification report, and confusion matrix\n# are omitted here as we don't have ground truth labels for the test set.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T11:04:48.909598Z","iopub.execute_input":"2025-04-29T11:04:48.909978Z","iopub.status.idle":"2025-04-29T11:04:49.308793Z","shell.execute_reply.started":"2025-04-29T11:04:48.909956Z","shell.execute_reply":"2025-04-29T11:04:49.308077Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 10. Generate Submission File ---\nimport pandas as pd # Ensure pandas is imported\n\nprint(\"\\n--- Generating Submission File ---\")\n\n# final_results_df was created in the previous cell\n\nprint(f\"Number of predictions in final_results_df: {len(final_results_df)}\")\nprint(f\"Number of unique id_codes in final_results_df: {final_results_df['id_code'].nunique()}\")\n\nif len(final_results_df) != final_results_df['id_code'].nunique():\n     print(\"Warning: Duplicate id_codes found in the combined results. Check concatenation logic.\")\n     # Optional: Decide how to handle duplicates, e.g., keep first\n     final_results_df = final_results_df.drop_duplicates(subset='id_code', keep='first')\n     print(f\"Dropped duplicates, new count: {len(final_results_df)}\")\n\n# Load the sample submission file to get all required id_codes\ntry:\n    sample_sub = pd.read_csv('../input/aptos2019-blindness-detection/sample_submission.csv')\n    print(f\"Sample submission length: {len(sample_sub)}\")\nexcept FileNotFoundError:\n    print(\"Error: sample_submission.csv not found. Cannot generate submission file correctly.\")\n    # Handle error appropriately, maybe stop execution\n    raise\n\n# Merge predictions with the sample submission using 'id_code'\n# Use a left merge to keep all IDs from the sample submission and match predictions\nsubmission_df = pd.merge(sample_sub[['id_code']], final_results_df, on='id_code', how='left')\n\n# Check for missing predictions (images potentially skipped during loading/processing)\nmissing_preds = submission_df['diagnosis'].isnull().sum()\nif missing_preds > 0:\n    print(f\"Warning: {missing_preds} images from sample_submission were not found in the processed results.\")\n    # Fill missing predictions. A common strategy is to predict 0 (No DR) for missing ones.\n    print(\"Filling missing predictions with 0 (No DR).\")\n    submission_df['diagnosis'] = submission_df['diagnosis'].fillna(0) # Fill NaN with 0\n    # Verify NaNs are filled\n    if submission_df['diagnosis'].isnull().sum() > 0:\n        print(\"Error: Failed to fill all missing predictions!\")\n\n\n# Ensure the diagnosis column is integer type\n# Use .astype(int) after filling NaNs\nsubmission_df['diagnosis'] = submission_df['diagnosis'].astype(int)\n\n\nprint(\"\\nFinal submission DataFrame head:\")\nprint(submission_df.head())\nprint(f\"\\nFinal submission DataFrame length: {len(submission_df)}\")\n\n# Check if final submission length matches sample submission\nif len(submission_df) != len(sample_sub):\n    print(f\"Error: Final submission length ({len(submission_df)}) does not match sample submission length ({len(sample_sub)})!\")\nelse:\n    print(\"Final submission length matches sample submission length.\")\n\n# Save the submission file\ntry:\n    submission_df.to_csv('submission.csv', index=False)\n    print(\"\\nSubmission file 'submission.csv' created successfully.\")\nexcept Exception as e:\n    print(f\"\\nError saving submission file: {e}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T11:04:49.309551Z","iopub.execute_input":"2025-04-29T11:04:49.309737Z","iopub.status.idle":"2025-04-29T11:04:49.351890Z","shell.execute_reply.started":"2025-04-29T11:04:49.309722Z","shell.execute_reply":"2025-04-29T11:04:49.351364Z"}},"outputs":[],"execution_count":null}]}