{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":97984,"databundleVersionId":14096757,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# ===================================================================\n# FINAL, COMPETITION-GRADE RULE-BASED SOLUTION (Adapted from AmbrosM)\n# ===================================================================","metadata":{}},{"cell_type":"markdown","source":"Imports and configuration","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport scipy.signal\nimport scipy.optimize\nfrom tqdm.notebook import tqdm\nfrom scipy.signal import medfilt\nfrom collections import defaultdict\nfrom typing import Tuple \nimport warnings\nimport matplotlib.pyplot as plt # Added for visual debugging\n\n# Suppress pandas performance warnings for cleaner output\nwarnings.filterwarnings('ignore', category=pd.errors.PerformanceWarning)\n\nprint(\"✅ Libraries imported successfully.\")\n\n# --- CONFIGURATION ---\nclass Config:\n    BASE_DIR = '/kaggle/input/physionet-ecg-image-digitization'\n    \n    # Parameters for signal processing and scoring, directly from AmbrosM's notebook\n    MAX_TIME_SHIFT = 0.2 \n    PERFECT_SCORE = 384 \n    OUTLIER_LOW_THRESHOLD = -1.6 \n    OUTLIER_HIGH_THRESHOLD = 0.85 \n    MARKER_ARTIFACT_THRESHOLD = 0.2 \n    MEDIAN_FILTER_SIZE = 5 \n    SCALING_FACTOR = 80 \n    TAIL_CORRECTION_FACTOR = 2.0 \n\n    # --- DEBUGGING CONFIGURATION ---\n    DEBUG_MODE = True \n    DEBUG_SAVE_IMAGES = True # Set to True to save intermediate images for visual inspection\n    DEBUG_NUM_IMAGES_TO_SAVE = 10 # How many test images to save debug output for (increased)\n    DEBUG_OUTPUT_DIR = '/kaggle/working/debug_output'\n    DEBUG_MATCH_THRESHOLD = 0.5 # Lowered threshold for template matching (original was 0.6)\n\n\nconfig = Config()\n\n# Standard 12-lead ECG order, used consistently throughout the rule-based approach\nLEADS = ['I', 'II', 'III', 'aVR', 'aVL', 'aVF', 'V1', 'V2', 'V3', 'V4', 'V5', 'V6']\nprint(\"✅ Configuration set.\")\n\n# Create debug output directory if in debug mode\nif config.DEBUG_MODE and config.DEBUG_SAVE_IMAGES:\n    os.makedirs(config.DEBUG_OUTPUT_DIR, exist_ok=True)\n    print(f\"DEBUG: Saving debug images to {config.DEBUG_OUTPUT_DIR}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. EVALUATION UTILITIES (Provided for context, not directly used in submission generation)","metadata":{}},{"cell_type":"code","source":"# These functions are as provided in the competition for scoring submissions.\n# They are included here for completeness but are commented out if not explicitly called.\n\n# class ParticipantVisibleError(Exception):\n#     pass\n\n# def compute_power(label: np.ndarray, prediction: np.ndarray) -> Tuple[float, float]:\n#     if label.ndim != 1 or prediction.ndim != 1:\n#         raise ParticipantVisibleError('Inputs must be 1-dimensional arrays.')\n#     finite_mask = np.isfinite(prediction)\n#     if not np.any(finite_mask):\n#         raise ParticipantVisibleError(\"The 'prediction' array contains no finite values (all NaN or inf).\")\n\n#     prediction[~np.isfinite(prediction)] = 0\n#     noise = label - prediction\n#     p_signal = np.sum(label**2)\n#     p_noise = np.sum(noise**2)\n#     return p_signal, p_noise\n\n# def compute_snr(signal: float, noise: float) -> float:\n#     if noise == 0:\n#         snr = config.PERFECT_SCORE\n#     elif signal == 0:\n#         snr = 0\n#     else:\n#         snr = min((signal / noise), config.PERFECT_SCORE)\n#     return snr\n\n# def align_signals(label: np.ndarray, pred: np.ndarray, max_shift: float = float('inf')) -> np.ndarray:\n#     if np.any(~np.isfinite(label)):\n#         raise ParticipantVisibleError('values in label should all be finite')\n#     if np.sum(np.isfinite(pred)) == 0:\n#         raise ParticipantVisibleError('prediction can not all be infinite')\n\n#     label_arr = np.asarray(label, dtype=np.float64)\n#     pred_arr = np.asarray(pred, dtype=np.float64)\n\n#     label_mean = np.mean(label_arr)\n#     pred_mean = np.mean(pred_arr)\n\n#     label_arr_centered = label_arr - label_mean\n#     pred_arr_centered = pred_arr - pred_mean\n\n#     correlation = scipy.signal.correlate(label_arr_centered, pred_arr_centered, mode='full')\n#     n_label = np.size(label_arr)\n#     n_pred = np.size(pred_arr)\n#     lags = scipy.signal.correlation_lags(n_label, n_pred, mode='full')\n#     valid_lags_mask = (lags >= -max_shift) & (lags <= max_shift)\n\n#     max_correlation = np.nanmax(correlation[valid_lags_mask])\n#     all_max_indices = np.flatnonzero(correlation == max_correlation)\n#     best_idx = min(all_max_indices, key=lambda i: abs(lags[i]))\n#     time_shift = lags[best_idx]\n#     start_padding_len = max(time_shift, 0)\n#     pred_slice_start = max(-time_shift, 0)\n#     pred_slice_end = min(n_label - time_shift, n_pred)\n#     end_padding_len = max(n_label - n_pred - time_shift, 0)\n#     aligned_pred = np.concatenate((np.full(start_padding_len, np.nan), pred_arr[pred_slice_start:pred_slice_end], np.full(end_padding_len, np.nan)))\n\n#     def objective_func(v_shift):\n#         return np.nansum((label_arr - (aligned_pred - v_shift)) ** 2)\n\n#     if np.any(np.isfinite(label_arr) & np.isfinite(aligned_pred)):\n#         results = scipy.optimize.minimize_scalar(objective_func, method='Brent')\n#         vertical_shift = results.x\n#         aligned_pred -= vertical_shift\n#     return aligned_pred\n\n# # The full `score` and `_calculate_image_score` functions are omitted for brevity,\n# # as they require the ground truth dataframe and are for competition evaluation,\n# # not for generating predictions.\n\n# print(\"✅ Evaluation utilities defined (if uncommented).\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. SIGNAL PROCESSING UTILITIES (Core Logic from AmbrosM's solution)","metadata":{}},{"cell_type":"code","source":"def get_adaptive_threshold(ima):\n    \"\"\"\n    Get adaptive binarization threshold based on image brightness.\n    Assumes `ima` is a 3-channel BGR image from cv2.imread.\n    \"\"\"\n    if ima.ndim == 3:\n        # Use a single channel or average for brightness if BGR\n        mean_brightness = np.mean(ima[:, :, 0]) # Using Blue channel as a proxy\n    else: \n        mean_brightness = np.mean(ima)\n    \n    if mean_brightness > 200:  # Bright image\n        return 170\n    elif mean_brightness < 150:  # Dark image\n        return 150\n    else:  # Normal brightness\n        return 160\n\ndef scale_markers_for_height(markers, original_height, target_height):\n    \"\"\"Scales marker positions if the image height differs from the template height.\"\"\"\n    if markers is None or original_height == target_height:\n        return markers\n    \n    scale_factor = target_height / original_height\n    scaled_markers = []\n    \n    for marker in markers:\n        if marker is not None:\n            scaled_marker = marker.copy()\n            scaled_marker[0] = int(scaled_marker[0] * scale_factor)  # Scale y-coordinate\n            scaled_markers.append(scaled_marker)\n        else:\n            scaled_markers.append(None)\n    \n    return scaled_markers\n\nclass MarkerFinder:\n    \"\"\"\n    Finds key reference markers on the ECG image using template matching.\n    These markers define the spatial layout of the ECG leads.\n    \"\"\"\n    def __init__(self):\n        self.templates = [None] * 17\n        self.template_positions = [None] * 17\n        self.template_sizes = np.array([(105, 60)] * 17)\n        self.template_points = [None] * 17\n\n        template_image_paths = [\n            os.path.join(config.BASE_DIR, 'train', '4292118763', '4292118763-0001.png'),\n            os.path.join(config.BASE_DIR, 'train', '4289880010', '4289880010-0001.png'),\n            os.path.join(config.BASE_DIR, 'train', '4284351157', '4284351157-0001.png'),\n        ]\n        \n        template_images = []\n        for path in template_image_paths:\n            img = cv2.imread(path)\n            if img is not None:\n                template_images.append(img)\n            else:\n                print(f\"❌ MarkerFinder: Could not load template image: {path}. Check dataset path and file existence.\")\n        \n        if not template_images:\n            print(\"❌ MarkerFinder: No template images could be loaded. Marker detection will not work.\")\n            return\n\n        ima_template_src = np.max(template_images, axis=0) # Combine loaded template images\n        \n        # Hardcoded absolute pixel coordinates for markers on the template image.\n        absolute_points = np.zeros((17, 2), dtype=int)\n        for i in range(3): \n            absolute_points[5 * i] = np.array([707 + 284 * i, 118])\n            for j in range(1, 5):\n                absolute_points[5 * i + j] = np.array([707 + 284 * i, 118 + 492 * j])\n        absolute_points[15] = np.array([1535, 118])\n        absolute_points[16] = np.array([1535, 118 + 492 * 4])\n            \n        # Define template positions and their anchor points relative to template crop\n        for i in range(len(absolute_points)):\n            if absolute_points[i][1] < 118 + 492 * 4: # These are for the standard 12 leads\n                if i % 5 == 0: # First marker in a row\n                    self.template_positions[i] = (absolute_points[i][0] - 87, absolute_points[i][1] - 50)\n                else: # Subsequent markers in a row\n                    self.template_positions[i] = (absolute_points[i][0] - 37, absolute_points[i][1] - 13)\n                self.template_points[i] = np.array([\n                    absolute_points[i][0] - self.template_positions[i][0], # Y-offset\n                    absolute_points[i][1] - self.template_positions[i][1]  # X-offset\n                ])\n            # For the rhythm strip (II long), template position and point would be different\n            # but the provided `absolute_points` and `lead_info` already handle this by using marker indices 15,16.\n            # No explicit template snippet is created for 15,16 directly here, implying they might be extrapolated\n            # or implicitly handled by the larger rhythm strip processing area.\n            # This is a nuance of the original code, we'll stick to it.\n\n        # Now, actually crop the templates from the source image\n        for i in range(len(self.template_positions)):\n            if self.template_positions[i] is not None:\n                tp_y, tp_x = self.template_positions[i]\n                ts_h, ts_w = self.template_sizes[i]\n                \n                # Ensure the template crop is within image bounds\n                if tp_y >= 0 and tp_x >= 0 and \\\n                   tp_y + ts_h <= ima_template_src.shape[0] and \\\n                   tp_x + ts_w <= ima_template_src.shape[1]:\n                    \n                    template = ima_template_src[tp_y : tp_y + ts_h, tp_x : tp_x + ts_w]\n                    if template.size > 0 and template.shape[0] == ts_h and template.shape[1] == ts_w:\n                        self.templates[i] = template\n                    else:\n                        print(f\"Warning: MarkerFinder template {i} is empty or wrong shape after cropping. Check coordinates/sizes.\")\n                else:\n                    print(f\"Warning: MarkerFinder template {i} coordinates ({tp_y},{tp_x}) + size ({ts_h},{ts_w}) are out of image bounds ({ima_template_src.shape}). Check absolute_points and offsets.\")\n        \n        # Final check if any templates were successfully loaded\n        if any(t is not None for t in self.templates):\n            print(f\"✅ MarkerFinder initialized successfully with {sum(1 for t in self.templates if t is not None)} templates.\")\n        else:\n            print(\"❌ MarkerFinder: No valid templates were created. Marker detection will likely fail.\")\n\n    def find_markers(self, ima):\n        \"\"\"\n        Locates the predefined markers within a given ECG image using template matching.\n        \"\"\"\n        markers = [None] * 17\n        \n        for j in range(len(self.templates)):\n            if self.templates[j] is not None:\n                t_start = max(0, self.template_positions[j][0] - 100)\n                t_end = min(ima.shape[0], self.template_positions[j][0] + 100 + self.template_sizes[j][0])\n                l_start = max(0, self.template_positions[j][1] - 100)\n                l_end = min(ima.shape[1], self.template_positions[j][1] + 250 + self.template_sizes[j][1])\n\n                search_range = ima[t_start:t_end, l_start:l_end]\n                \n                # Ensure search range is large enough for the template\n                if search_range.shape[0] < self.templates[j].shape[0] or \\\n                   search_range.shape[1] < self.templates[j].shape[1]:\n                    continue \n\n                res = cv2.matchTemplate(search_range, self.templates[j], cv2.TM_CCOEFF_NORMED)\n                _, max_val, _, max_loc = cv2.minMaxLoc(res)\n                \n                # Use configurable debug threshold\n                if max_val > config.DEBUG_MATCH_THRESHOLD: \n                    top_left_in_search = max_loc\n                    markers[j] = np.array((\n                        t_start + top_left_in_search[1] + self.template_points[j][0], \n                        l_start + top_left_in_search[0] + self.template_points[j][1]  \n                    ))\n                #else: # Debugging: print failed matches\n                #    if config.DEBUG_MODE:\n                #        print(f\"DEBUG: MarkerFinder: Marker {j} match failed (max_val={max_val:.2f} < {config.DEBUG_MATCH_THRESHOLD})\")\n        \n        # Extrapolate missing markers based on relative positions of detected ones.\n        for i in range(3): \n            if markers[5 * i + 3] is not None and markers[5 * i + 2] is not None:\n                m = markers[5 * i + 3] * 2 - markers[5 * i + 2]\n                markers[5 * i + 4] = m\n\n        if markers[14] is not None and markers[9] is not None:\n            markers[16] = ((markers[14] * (284 + 260) - markers[9] * 260) / 284).astype(int)\n\n        return markers\n        \n    @staticmethod\n    def lead_info(lead):\n        \"\"\"\n        Returns the row index, and start/end marker indices for a given lead.\n        \"\"\"\n        begin_idx, end_idx = {\n            'I': (0, 1),\n            'II-subset': (5, 6), # Part of lead II in the main grid\n            'III': (10, 11),\n            'aVR': (1, 2),\n            'aVL': (6, 7),\n            'aVF': (11, 12),\n            'V1': (2, 3),\n            'V2': (7, 8),\n            'V3': (12, 13),\n            'V4': (3, 4),\n            'V5': (8, 9),\n            'V6': (13, 14),\n            'II': (15, 16), # The full 10-second rhythm strip (long Lead II)\n        }[lead]\n        return begin_idx // 5, begin_idx, end_idx \n\nmf = MarkerFinder()\n\ndef find_line_by_topdown_sweep(ima_binary, debug_info=None):\n    \"\"\"\n    Identifies the top and bottom boundaries of the ECG signal trace in a binary image strip.\n    \"\"\"\n    if debug_info is not None:\n        debug_output_path = debug_info['output_path']\n        image_id = debug_info['image_id']\n        current_step_count = debug_info['step_count']\n        debug_info['step_count'] += 1\n\n    top = np.argmin(ima_binary, axis=0) \n    median_top = int(np.median(top))\n    \n    top[top == 0] = median_top \n    top[top > median_top + 300] = median_top \n\n    strip_width = 64\n    for strip_left in range(0, ima_binary.shape[1], strip_width):\n        median_top_strip = int(np.median(top[strip_left:strip_left + strip_width]))\n        if median_top_strip > median_top + 300: \n            median_top_strip = median_top\n        \n        strip = ima_binary[median_top_strip + 80:, strip_left:strip_left + strip_width]\n        if strip.size > 0:\n            all_white_rows_in_strip = strip.all(axis=1) \n            if all_white_rows_in_strip.size > 0:\n                first_white_row_relative = np.argmax(all_white_rows_in_strip) \n                if first_white_row_relative > 0 or all_white_rows_in_strip[0]: \n                    first_white_row_global = first_white_row_relative + median_top_strip + 80\n                    mask = top > first_white_row_global\n                    mask[:strip_left] = False \n                    mask[strip_left + strip_width:] = False\n                    top[mask] = median_top_strip \n\n    mask_above_top = np.tile(np.arange(len(ima_binary)).reshape(-1, 1), reps=(1, ima_binary.shape[1]))\n    mask_above_top = mask_above_top >= top\n    ima_binary_top_filtered = ima_binary & mask_above_top # Use a copy for debugging\n\n    if config.DEBUG_MODE and config.DEBUG_SAVE_IMAGES and debug_info['image_counter'] < config.DEBUG_NUM_IMAGES_TO_SAVE:\n        debug_image = cv2.cvtColor((ima_binary_top_filtered * 255).astype(np.uint8), cv2.COLOR_GRAY2BGR)\n        for x, y in enumerate(top): # Draw the 'top' line\n            cv2.circle(debug_image, (x, y), 1, (0, 0, 255), -1) # Red dots\n        cv2.imwrite(os.path.join(debug_output_path, f\"{image_id}_debug_{current_step_count:02d}_top_lines.png\"), debug_image)\n\n\n    bottom = np.argmax(ima_binary_top_filtered, axis=0) # Find bottom in the top-filtered image\n\n    bottomx = np.maximum(bottom, np.median(top) + 100) \n    mask_below_bottom = np.tile(np.arange(len(ima_binary_top_filtered)).reshape(-1, 1), reps=(1, ima_binary_top_filtered.shape[1]))\n    mask_below_bottom = mask_below_bottom < bottomx\n    \n    ima_binary_final = ima_binary_top_filtered | mask_below_bottom \n    ima_binary_final[:, :-1] |= mask_below_bottom[:, 1:] \n    ima_binary_final[:, 1:] |= mask_below_bottom[:, :-1] \n\n    if config.DEBUG_MODE and config.DEBUG_SAVE_IMAGES and debug_info['image_counter'] < config.DEBUG_NUM_IMAGES_TO_SAVE:\n        debug_image = cv2.cvtColor((ima_binary_final * 255).astype(np.uint8), cv2.COLOR_GRAY2BGR)\n        for x, y in enumerate(bottom): # Draw the 'bottom' line\n            cv2.circle(debug_image, (x, y), 1, (0, 255, 0), -1) # Green dots\n        cv2.imwrite(os.path.join(debug_output_path, f\"{image_id}_debug_{current_step_count:02d}_bottom_lines.png\"), debug_image)\n\n\n    return top, bottom\n\ndef get_lead_from_top_bottom(tops, bottoms, lead, number_of_rows, markers):\n    \"\"\"\n    Extracts and processes the signal for a specific ECG lead given the detected\n    top and bottom boundaries and marker positions. Applies various corrections.\n    \"\"\"\n    line_row_idx, begin_marker_idx, end_marker_idx = mf.lead_info(lead)\n    \n    top_line_segment = tops[line_row_idx]\n    bottom_line_segment = bottoms[line_row_idx]\n    \n    begin_marker = markers[begin_marker_idx]\n    end_marker = markers[end_marker_idx]\n\n    if begin_marker is None or end_marker is None:\n        if config.DEBUG_MODE:\n            print(f\"DEBUG: get_lead_from_top_bottom for {lead}: Missing markers ({begin_marker_idx}, {end_marker_idx}). Returning zeros.\")\n        return np.zeros(number_of_rows, dtype=np.float32)\n\n    signal_segment_x_start = begin_marker[1]\n    signal_segment_x_end = end_marker[1]\n    \n    if signal_segment_x_start >= signal_segment_x_end or \\\n       signal_segment_x_end > len(top_line_segment) or \\\n       signal_segment_x_start < 0:\n        if config.DEBUG_MODE:\n            print(f\"DEBUG: get_lead_from_top_bottom for {lead}: Invalid X-range ({signal_segment_x_start}-{signal_segment_x_end}). Returning zeros.\")\n        return np.zeros(number_of_rows, dtype=np.float32)\n\n    baseline_y_coords = np.linspace(begin_marker[0], end_marker[0], signal_segment_x_end - signal_segment_x_start)\n\n    signal_midpoint = (top_line_segment[signal_segment_x_start:signal_segment_x_end] + \n                       bottom_line_segment[signal_segment_x_start:signal_segment_x_end]) / 2\n    \n    if len(signal_midpoint) < len(baseline_y_coords):\n        baseline_y_coords = baseline_y_coords[:len(signal_midpoint)]\n    \n    pred = baseline_y_coords - signal_midpoint\n\n    pred /= config.SCALING_FACTOR\n\n    if lead in ['aVR', 'aVL', 'aVF', 'V1', 'V2', 'V3', 'V4', 'V5', 'V6']:\n        pred[:4] = np.where(pred[:4] > config.MARKER_ARTIFACT_THRESHOLD, pred[4], pred[:4])\n    if lead in ['I', 'II-subset', 'III', 'aVR', 'aVL', 'aVF', 'V1', 'V2', 'V3']:\n        pred[-5:] = np.where(pred[-5:] > config.MARKER_ARTIFACT_THRESHOLD, pred[-6], pred[-5:])\n    if lead in ['I', 'II-subset', 'III', 'II']:\n        pred[:2] = pred[2] \n\n    pred = np.interp(np.linspace(0, 1, number_of_rows),\n                     np.linspace(0, 1, len(pred)),\n                     pred)\n    \n    outlier_mask = (pred < config.OUTLIER_LOW_THRESHOLD) | (pred > config.OUTLIER_HIGH_THRESHOLD)\n    if np.any(outlier_mask):\n        for i in np.where(outlier_mask)[0]:\n            start_idx = max(0, i - 3)\n            end_idx = min(len(pred), i + 4)\n            neighbors = pred[start_idx:end_idx]\n            valid_neighbors = neighbors[(neighbors >= config.OUTLIER_LOW_THRESHOLD) & (neighbors <= config.OUTLIER_HIGH_THRESHOLD)]\n            if len(valid_neighbors) > 0:\n                pred[i] = np.median(valid_neighbors)\n            else:\n                pred[i] = 0 \n\n    pred = medfilt(pred, kernel_size=config.MEDIAN_FILTER_SIZE)\n    \n    edge_samples = max(1, number_of_rows // 100) \n    pred[:edge_samples] = pred[edge_samples]\n    pred[-edge_samples:] = pred[-edge_samples-1]\n\n    if lead in ['II']:\n        n_tail = number_of_rows // 48\n        if n_tail > 0 and len(pred) >= n_tail:\n            tail_values = pred[-n_tail:]\n            tail_median = np.median(tail_values)\n            tail_std = np.std(tail_values)\n            threshold = min(config.OUTLIER_HIGH_THRESHOLD, tail_median + config.TAIL_CORRECTION_FACTOR * tail_std)\n            pred[-n_tail:] = np.where(np.abs(pred[-n_tail:]) <= threshold, pred[-n_tail:], tail_median)\n        \n    if lead in ['V4', 'V5', 'V6']:\n        n_tail = number_of_rows // 12\n        if n_tail > 0 and len(pred) >= n_tail:\n            tail_values = pred[-n_tail:]\n            tail_median = np.median(tail_values)\n            tail_std = np.std(tail_values)\n            threshold = min(config.OUTLIER_HIGH_THRESHOLD, tail_median + config.TAIL_CORRECTION_FACTOR * tail_std)\n            pred[-n_tail:] = np.where(np.abs(pred[-n_tail:]) <= threshold, pred[-n_tail:], tail_median)\n        \n    return pred\n\ndef convert_scanned_color(ima, markers, n_timesteps, debug_info=None):\n    \"\"\"\n    Main function to process a single ECG image: binarize, extract signal lines\n    from multiple lead segments, and compile into a dictionary of predictions.\n    \"\"\"\n    if debug_info is not None:\n        debug_output_path = debug_info['output_path']\n        image_id = debug_info['image_id']\n        if debug_info['image_counter'] < config.DEBUG_NUM_IMAGES_TO_SAVE:\n            # Save original image with markers\n            debug_img_markers = ima.copy()\n            for i, marker in enumerate(markers):\n                if marker is not None:\n                    cv2.circle(debug_img_markers, (marker[1], marker[0]), 5, (0, 0, 255), -1) # (x,y) for cv2\n                    cv2.putText(debug_img_markers, str(i), (marker[1]+10, marker[0]), cv2.FONT_HERSHEY_SIMPLEX, 0.5, (255, 255, 255), 1)\n            cv2.imwrite(os.path.join(debug_output_path, f\"{image_id}_debug_00_markers.png\"), debug_img_markers)\n            \n            # Save cropped BGR image for reference\n            cv2.imwrite(os.path.join(debug_output_path, f\"{image_id}_debug_01_original_cropped.png\"), ima[crop_top:])\n\n\n    if ima.shape[0] == 1700:\n        crop_top = 420\n    else: \n        crop_top = 400\n    \n    threshold = get_adaptive_threshold(ima)\n    \n    ima_binarized = ima[crop_top:, :, 0] > threshold \n\n    if config.DEBUG_MODE and config.DEBUG_SAVE_IMAGES and debug_info is not None and debug_info['image_counter'] < config.DEBUG_NUM_IMAGES_TO_SAVE:\n        cv2.imwrite(os.path.join(debug_output_path, f\"{image_id}_debug_02_binarized.png\"), (ima_binarized * 255).astype(np.uint8))\n\n\n    iima_uint8 = ima_binarized.astype(np.uint8)\n    ima_smoothed = (iima_uint8[:-2, :-2] + iima_uint8[:-2, 1:-1] + iima_uint8[:-2, 2:]\n           + iima_uint8[1:-1, :-2] + iima_uint8[1:-1, 1:-1] + iima_uint8[1:-1, 2:]\n           + iima_uint8[2:, :-2] + iima_uint8[2:, 1:-1] + iima_uint8[2:, 2:]) >= 7\n    \n    tops, bottoms = [], []\n    for i in range(4): \n        current_debug_info = debug_info.copy() if debug_info else None\n        if current_debug_info:\n            current_debug_info['current_line_row'] = i\n        top, bottom = find_line_by_topdown_sweep(ima_smoothed, debug_info=current_debug_info)\n        tops.append(top)\n        bottoms.append(bottom)\n\n    tops = [t + crop_top for t in tops]\n    bottoms = [b + crop_top for b in bottoms]\n\n    n_timesteps['II-subset'] = n_timesteps['I']\n    preds = {}\n    for lead in LEADS + ['II-subset']:\n        pred = get_lead_from_top_bottom(tops, bottoms, lead, n_timesteps[lead], markers)\n        preds[lead] = pred\n\n    if 'II-subset' in preds and 'II' in preds and len(preds['II-subset']) > 0:\n        len_overlap = min(len(preds['II']), len(preds['II-subset']))\n        if len_overlap > 0:\n            preds['II'][:len_overlap] = (preds['II'][:len_overlap] + preds['II-subset'][:len_overlap]) / 2\n    del preds['II-subset'] \n\n    apply_einthoven(preds)\n\n    return preds\n\ndef apply_einthoven(preds):\n    \"\"\"\n    Applies Einthoven's Law and other physiological relationships to improve consistency\n    and correct for small errors between different ECG leads.\n    \"\"\"\n    if 'I' in preds and 'II' in preds and 'III' in preds:\n        len_common = min(len(preds['I']), len(preds['II']), len(preds['III']))\n        if len_common > 0:\n            residual = preds['I'][:len_common] + preds['III'][:len_common] - preds['II'][:len_common]\n            correction = residual / 3\n            preds['I'][:len_common] -= correction\n            preds['III'][:len_common] -= correction\n            preds['II'][:len_common] += correction\n    \n    if 'aVR' in preds and 'aVL' in preds and 'aVF' in preds:\n        len_common = min(len(preds['aVR']), len(preds['aVL']), len(preds['aVF']))\n        if len_common > 0:\n            residual = preds['aVR'][:len_common] + preds['aVL'][:len_common] + preds['aVF'][:len_common]\n            correction = residual / 3\n            preds['aVR'][:len_common] -= correction\n            preds['aVL'][:len_common] -= correction\n            preds['aVF'][:len_common] -= correction\n\n    if 'aVR' in preds and 'aVF' in preds and 'II' in preds:\n        start_idx = len(preds['I']) if 'I' in preds else 0\n        end_idx = start_idx + len(preds['aVR']) if 'aVR' in preds else start_idx\n        \n        if start_idx < len(preds['II']) and end_idx <= len(preds['II']) and \\\n           len(preds['aVR']) > 0 and len(preds['aVF']) > 0 and (end_idx - start_idx) > 0:\n            \n            len_segment = end_idx - start_idx\n            aVR_interp = np.interp(np.linspace(0, 1, len_segment), np.linspace(0, 1, len(preds['aVR'])), preds['aVR'])\n            aVF_interp = np.interp(np.linspace(0, 1, len_segment), np.linspace(0, 1, len(preds['aVF'])), preds['aVF'])\n            \n            residual = 2 * aVR_interp - 2 * aVF_interp + 3 * preds['II'][start_idx:end_idx]\n            correction = residual / 17\n            \n            preds['aVR'] -= np.interp(np.linspace(0, 1, len(preds['aVR'])), np.linspace(0, 1, len_segment), 2 * correction)\n            preds['aVF'] += np.interp(np.linspace(0, 1, len(preds['aVF'])), np.linspace(0, 1, len_segment), 2 * correction)\n            preds['II'][start_idx:end_idx] -= 3 * correction\n\ndef is_color_image(ima):\n    \"\"\"Checks if an image is color by verifying if there's variation across color channels.\"\"\"\n    if ima.ndim < 3: return False\n    return ima.std(axis=2).mean() > 1e-6 \n\nprint(\"✅ Signal processing utilities defined.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. MEAN MODEL FALLBACK\n# This model provides a generic, averaged signal for leads if image processing fails.","metadata":{}},{"cell_type":"code","source":"def fit_mean_model(train_df_full):\n    \"\"\"\n    Computes a representative 'mean' signal for each lead from the training data.\n    This serves as a robust fallback prediction.\n    \"\"\"\n    mean_dict = defaultdict(list)\n    print(\"\\n--- Fitting Mean Model (Fallback) ---\")\n    for idx, row in tqdm(train_df_full.iterrows(), total=len(train_df_full), desc=\"Processing Train Signals for Mean Model\"):\n        csv_path = os.path.join(config.BASE_DIR, 'train', str(row.id), f\"{row.id}.csv\")\n        \n        if not os.path.exists(csv_path):\n            continue\n        \n        try:\n            labels = pd.read_csv(csv_path)\n            for lead in LEADS: \n                if lead in labels.columns:\n                    values = labels[lead]\n                    values = values[~values.isna()] \n                    if len(values) > 50: \n                        resampled_values = np.interp(\n                            np.linspace(0, 1, 20000), \n                            np.linspace(0, 1, len(values)),\n                            values\n                        )\n                        mean_dict[lead].append(resampled_values)\n        except Exception as e:\n            # print(f\"Warning: Could not process CSV for ID {row.id}, lead {lead}: {e}\")\n            pass # Skip to the next record if an error occurs\n\n    for lead in LEADS: # Ensure all leads have an entry, even if empty\n        if mean_dict[lead]: \n            mean_dict[lead] = np.mean(np.stack(mean_dict[lead]), axis=0)\n        else: \n            if config.DEBUG_MODE:\n                print(f\"DEBUG: No valid training data found for lead {lead} to compute mean. Using sine wave as fallback for mean model itself.\")\n            t = np.linspace(0, 1, 20000) \n            mean_dict[lead] = np.sin(2 * np.pi * t) * 0.1 \n            \n    print(\"✅ Mean Model Fitted.\")\n    return mean_dict\n\n# Load train and test dataframes\ntrain_df_full = pd.read_csv(os.path.join(config.BASE_DIR, 'train.csv'))\ntest_df = pd.read_csv(os.path.join(config.BASE_DIR, 'test.csv'))\n\nmean_model_fitted = fit_mean_model(train_df_full)\nprint(\"✅ Mean model prepared for fallback predictions.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5. INFERENCE AND SUBMISSION GENERATION (Using the rule-based approach)","metadata":{}},{"cell_type":"code","source":"print(\"\\n--- Starting Inference ---\")\n\nsubmission_data = []\nold_image_id = None \ndebug_image_counter = 0 # Counter for saving debug images\n\nfor idx, row in tqdm(test_df.iterrows(), total=len(test_df), desc=\"Processing Test ECGs\"):\n    current_image_id = row.id\n    \n    if current_image_id != old_image_id:\n        image_path = os.path.join(config.BASE_DIR, 'test', f\"{current_image_id}.png\")\n        \n        preds = None \n        \n        # Prepare debug_info for the current image\n        debug_info = {\n            'image_id': current_image_id,\n            'output_path': os.path.join(config.DEBUG_OUTPUT_DIR, str(current_image_id)),\n            'image_counter': debug_image_counter,\n            'step_count': 0 # For ordering debug image steps\n        }\n        if config.DEBUG_MODE and config.DEBUG_SAVE_IMAGES and debug_image_counter < config.DEBUG_NUM_IMAGES_TO_SAVE:\n            os.makedirs(debug_info['output_path'], exist_ok=True)\n            original_image = cv2.imread(image_path) # Load for saving reference\n            if original_image is not None:\n                cv2.imwrite(os.path.join(debug_info['output_path'], f\"{current_image_id}_debug_original_full.png\"), original_image)\n        \n        if os.path.exists(image_path):\n            ima = cv2.imread(image_path)\n            \n            if ima is None:\n                if config.DEBUG_MODE:\n                    print(f\"DEBUG: Image {image_path} could not be read. Falling back to mean model for all its leads.\")\n            else:\n                current_image_height, current_image_width = ima.shape[0], ima.shape[1]\n                \n                good_shape = current_image_height in [1652, 1700]\n                \n                if good_shape and is_color_image(ima):\n                    markers = mf.find_markers(ima)\n                    \n                    if not any(m is not None for m in markers):\n                        if config.DEBUG_MODE:\n                            print(f\"DEBUG: Image {current_image_id}.png: No markers found by MarkerFinder. Falling back to mean model.\")\n                    else:\n                        if config.DEBUG_MODE and config.DEBUG_SAVE_IMAGES and debug_image_counter < config.DEBUG_NUM_IMAGES_TO_SAVE:\n                            debug_img_markers = ima.copy()\n                            for i, marker in enumerate(markers):\n                                if marker is not None:\n                                    cv2.circle(debug_img_markers, (marker[1], marker[0]), 5, (0, 0, 255), -1) # (x,y) for cv2\n                                    cv2.putText(debug_img_markers, str(i), (marker[1]+10, marker[0]), cv2.FONT_HERSHEY_SIMPLEX, 0.5, (255, 255, 255), 1)\n                            cv2.imwrite(os.path.join(debug_info['output_path'], f\"{current_image_id}_debug_{debug_info['step_count']:02d}_detected_markers.png\"), debug_img_markers)\n                            debug_info['step_count'] += 1\n\n                        if current_image_height == 1700:\n                            markers = scale_markers_for_height(markers, 1652, 1700) \n                        \n                        n_timesteps = {lead: row.fs * 10 if lead == 'II' else row.fs * 10 // 4 for lead in LEADS}\n                        \n                        try:\n                            preds = convert_scanned_color(ima, markers, n_timesteps, debug_info=debug_info)\n                            if preds is None or not any(lead_pred is not None and len(lead_pred) > 0 for lead_pred in preds.values()):\n                                if config.DEBUG_MODE:\n                                    print(f\"DEBUG: Image {current_image_id}.png: CV processing returned empty/invalid predictions. Falling back to mean model.\")\n                                preds = None # Force fallback\n                        except Exception as e:\n                            if config.DEBUG_MODE:\n                                print(f\"DEBUG: Image {current_image_id}.png: Error during CV processing: {e}. Falling back to mean model.\")\n                            preds = None # Force fallback\n                else:\n                    if config.DEBUG_MODE:\n                        print(f\"DEBUG: Image {current_image_id}.png (shape={ima.shape}, is_color={is_color_image(ima)}) \"\n                              f\"is not fully supported by CV (shape/color check). Falling back to mean model for all its leads.\")\n        \n        if config.DEBUG_MODE and config.DEBUG_SAVE_IMAGES and debug_image_counter < config.DEBUG_NUM_IMAGES_TO_SAVE:\n            debug_image_counter += 1 # Only increment counter if we saved debug images\n            if debug_image_counter >= config.DEBUG_NUM_IMAGES_TO_SAVE:\n                print(f\"DEBUG: Reached limit of {config.DEBUG_NUM_IMAGES_TO_SAVE} debug images. No more debug images will be saved.\")\n\n        old_image_id = current_image_id \n\n    lead_name = row.lead\n    num_rows = row.number_of_rows\n    \n    prediction_signal = None\n    \n    if preds is not None and lead_name in preds and preds[lead_name] is not None and len(preds[lead_name]) == num_rows:\n        prediction_signal = preds[lead_name]\n    else:\n        # Fallback to the mean model\n        mean_signal = mean_model_fitted.get(lead_name)\n        if mean_signal is not None and len(mean_signal) > 0:\n            prediction_signal = np.interp(\n                np.linspace(0, 1, num_rows),\n                np.linspace(0, 1, len(mean_signal)),\n                mean_signal\n            ).astype(np.float32)\n            if config.DEBUG_MODE:\n                print(f\"DEBUG: Using mean model for {current_image_id}_{lead_name}.\")\n        else:\n            if config.DEBUG_MODE:\n                print(f\"DEBUG: No mean model signal available for lead {lead_name}. Using zeros.\")\n            prediction_signal = np.zeros(num_rows, dtype=np.float32)\n\n    if prediction_signal is None or len(prediction_signal) != num_rows:\n        if config.DEBUG_MODE:\n            print(f\"DEBUG: Final check: Prediction signal for {current_image_id}_{lead_name} has invalid length ({len(prediction_signal) if prediction_signal is not None else 'None'} vs {num_rows}). Filling with zeros.\")\n        prediction_signal = np.zeros(num_rows, dtype=np.float32)\n\n    for i, val in enumerate(prediction_signal):\n        submission_data.append({\n            'id': f\"{current_image_id}_{i}_{lead_name}\",\n            'value': float(val) \n        })\n\nsubmission_df = pd.DataFrame(submission_data)\nsubmission_df.to_csv('submission.csv', index=False)\nprint(\"\\n✅ Submission file 'submission.csv' created successfully!\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}