{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":113558,"databundleVersionId":14174843,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nfrom pathlib import Path\nimport os\nfrom tqdm import tqdm\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Set up plotting style\nplt.style.use('ggplot')\nsns.set_palette(\"husl\")\n\ndef decode_mask(mask_path, img_shape):\n    \"\"\"\n    Decode mask from various formats to a standard binary mask\n    \"\"\"\n    mask_data = np.load(mask_path, allow_pickle=True)\n    H, W = img_shape[:2]\n\n    # ---- (2, N) format → coordinate list ----\n    if mask_data.ndim == 2 and mask_data.shape[0] == 2:\n        y, x = mask_data.astype(int)\n        mask = np.zeros((H, W), dtype=np.uint8)\n        mask[y, x] = 1\n\n    # ---- (2, H, W) format → source and target masks ----\n    elif mask_data.ndim == 3 and mask_data.shape[0] == 2:\n        mask = np.clip(mask_data[0] + mask_data[1], 0, 1).astype(np.uint8)\n\n    # ---- (1, H, W) format → single binary channel ----\n    elif mask_data.ndim == 3 and mask_data.shape[0] == 1:\n        mask = (mask_data[0] > 0).astype(np.uint8)\n\n    # ---- (3, H, W) format → multi-channel (combine to one) ----\n    elif mask_data.ndim == 3 and mask_data.shape[0] == 3:\n        mask = np.max(mask_data, axis=0).astype(np.uint8)\n        mask = (mask > 0).astype(np.uint8)\n\n    # ---- (H, W) already binary ----\n    elif mask_data.ndim == 2:\n        mask = (mask_data > 0).astype(np.uint8)\n\n    else:\n        print(f\"Warning: Unknown mask shape: {mask_data.shape}\")\n        mask = np.zeros((H, W), dtype=np.uint8)\n\n    # Ensure same size as image (resize if needed)\n    if mask.shape != (H, W):\n        mask = cv2.resize(mask, (W, H), interpolation=cv2.INTER_NEAREST)\n\n    return mask\n\nclass ScientificImageEDA:\n    def __init__(self, data_dir):\n        self.data_dir = Path(data_dir)\n        self.train_authentic_dir = self.data_dir / 'train_images' / 'authentic'\n        self.train_forged_dir = self.data_dir / 'train_images' / 'forged'\n        self.train_masks_dir = self.data_dir / 'train_masks'\n        self.test_images_dir = self.data_dir / 'test_images'\n        \n    def explore_directory_structure(self):\n        \"\"\"Explore and print directory structure\"\"\"\n        print(\"📁 DIRECTORY STRUCTURE ANALYSIS\")\n        print(\"=\" * 50)\n        \n        directories = {\n            'Train Authentic': self.train_authentic_dir,\n            'Train Forged': self.train_forged_dir,\n            'Train Masks': self.train_masks_dir,\n            'Test Images': self.test_images_dir\n        }\n        \n        for name, path in directories.items():\n            if path.exists():\n                files = list(path.glob('*'))\n                print(f\"{name}: {len(files)} files\")\n                if files:\n                    print(f\"  Sample files: {[f.name for f in files[:3]]}\")\n                print(f\"  Extensions: {list(set(f.suffix for f in files))}\")\n            else:\n                print(f\"{name}: Directory not found\")\n            print(\"-\" * 30)\n    \n    def get_dataset_stats(self):\n        \"\"\"Get comprehensive dataset statistics\"\"\"\n        print(\"\\n📊 DATASET STATISTICS\")\n        print(\"=\" * 50)\n        \n        stats = {}\n        \n        # Count files\n        stats['authentic_images'] = len(list(self.train_authentic_dir.glob('*.png')))\n        stats['forged_images'] = len(list(self.train_forged_dir.glob('*.png')))\n        stats['mask_files'] = len(list(self.train_masks_dir.glob('*.npy')))\n        stats['test_images'] = len(list(self.test_images_dir.glob('*.png')))\n        \n        # Print basic stats\n        for key, value in stats.items():\n            print(f\"{key.replace('_', ' ').title()}: {value}\")\n        \n        print(f\"\\nTotal Training Images: {stats['authentic_images'] + stats['forged_images']}\")\n        print(f\"Forgery Ratio: {stats['forged_images']/(stats['authentic_images'] + stats['forged_images']):.2%}\")\n        \n        return stats\n    \n    def analyze_image_properties(self, sample_size=50):\n        \"\"\"Analyze image dimensions, channels, and data types\"\"\"\n        print(\"\\n🖼️ IMAGE PROPERTIES ANALYSIS\")\n        print(\"=\" * 50)\n        \n        # Collect sample images from all categories\n        all_images = []\n        categories = [\n            ('authentic', self.train_authentic_dir),\n            ('forged', self.train_forged_dir),\n            ('test', self.test_images_dir)\n        ]\n        \n        properties = []\n        for category, directory in categories:\n            if directory.exists():\n                image_files = list(directory.glob('*.png'))[:sample_size]\n                for img_path in tqdm(image_files, desc=f\"Analyzing {category}\"):\n                    try:\n                        img = cv2.imread(str(img_path))\n                        if img is not None:\n                            height, width, channels = img.shape\n                            properties.append({\n                                'category': category,\n                                'filename': img_path.stem,\n                                'height': height,\n                                'width': width,\n                                'channels': channels,\n                                'dtype': str(img.dtype),\n                                'min_val': img.min(),\n                                'max_val': img.max(),\n                                'mean_val': img.mean(),\n                                'std_val': img.std()\n                            })\n                    except Exception as e:\n                        print(f\"Error processing {img_path}: {e}\")\n        \n        self.img_properties_df = pd.DataFrame(properties)\n        \n        # Print summary statistics\n        print(\"\\nImage Dimensions Summary:\")\n        print(self.img_properties_df.groupby('category')[['height', 'width']].describe())\n        \n        return self.img_properties_df\n    \n    def analyze_mask_properties(self, sample_size=50):\n        \"\"\"Analyze mask properties and distributions\"\"\"\n        print(\"\\n🎭 MASK PROPERTIES ANALYSIS\")\n        print(\"=\" * 50)\n        \n        if not self.train_masks_dir.exists():\n            print(\"Masks directory not found!\")\n            return None\n        \n        mask_files = list(self.train_masks_dir.glob('*.npy'))[:sample_size]\n        mask_properties = []\n        \n        for mask_path in tqdm(mask_files, desc=\"Analyzing masks\"):\n            try:\n                # Get corresponding image to know the shape\n                img_filename = mask_path.stem + '.png'\n                img_path = self.train_forged_dir / img_filename\n                \n                if img_path.exists():\n                    img = cv2.imread(str(img_path))\n                    mask = decode_mask(mask_path, img.shape)\n                    \n                    unique_vals, counts = np.unique(mask, return_counts=True)\n                    \n                    mask_properties.append({\n                        'filename': mask_path.stem,\n                        'mask_shape': mask.shape,\n                        'original_mask_shape': np.load(mask_path).shape,\n                        'unique_values': len(unique_vals),\n                        'max_value': mask.max(),\n                        'total_pixels': mask.size,\n                        'forged_pixels': np.sum(mask > 0),\n                        'forgery_ratio': np.sum(mask > 0) / mask.size if mask.size > 0 else 0,\n                        'mask_format': str(np.load(mask_path).shape)\n                    })\n            except Exception as e:\n                print(f\"Error processing {mask_path}: {e}\")\n        \n        self.mask_properties_df = pd.DataFrame(mask_properties)\n        \n        print(\"\\nMask Properties Summary:\")\n        print(self.mask_properties_df.describe())\n        \n        # Analyze mask formats\n        print(\"\\n📋 MASK FORMATS DISTRIBUTION:\")\n        if not self.mask_properties_df.empty:\n            format_counts = self.mask_properties_df['mask_format'].value_counts()\n            for format_type, count in format_counts.items():\n                print(f\"  {format_type}: {count} masks\")\n        \n        return self.mask_properties_df\n    \n    def visualize_image_distributions(self):\n        \"\"\"Create visualizations for image property distributions\"\"\"\n        print(\"\\n📈 VISUALIZING DISTRIBUTIONS\")\n        \n        fig, axes = plt.subplots(2, 3, figsize=(18, 12))\n        fig.suptitle('Image Properties Distribution Analysis', fontsize=16, fontweight='bold')\n        \n        # 1. Image dimensions scatter plot\n        if hasattr(self, 'img_properties_df'):\n            categories = self.img_properties_df['category'].unique()\n            colors = {'authentic': 'green', 'forged': 'red', 'test': 'blue'}\n            \n            for category in categories:\n                subset = self.img_properties_df[self.img_properties_df['category'] == category]\n                axes[0,0].scatter(subset['width'], subset['height'], \n                                alpha=0.6, label=category, color=colors.get(category, 'gray'))\n            \n            axes[0,0].set_xlabel('Width')\n            axes[0,0].set_ylabel('Height')\n            axes[0,0].set_title('Image Dimensions Scatter Plot')\n            axes[0,0].legend()\n            axes[0,0].grid(True)\n            \n            # 2. Aspect ratio distribution\n            self.img_properties_df['aspect_ratio'] = self.img_properties_df['width'] / self.img_properties_df['height']\n            for category in categories:\n                subset = self.img_properties_df[self.img_properties_df['category'] == category]\n                axes[0,1].hist(subset['aspect_ratio'], alpha=0.7, label=category, \n                              bins=20, color=colors.get(category, 'gray'))\n            \n            axes[0,1].set_xlabel('Aspect Ratio (Width/Height)')\n            axes[0,1].set_ylabel('Frequency')\n            axes[0,1].set_title('Aspect Ratio Distribution')\n            axes[0,1].legend()\n            \n            # 3. Intensity distributions\n            for category in categories:\n                subset = self.img_properties_df[self.img_properties_df['category'] == category]\n                axes[0,2].hist(subset['mean_val'], alpha=0.7, label=category, \n                              bins=20, color=colors.get(category, 'gray'))\n            \n            axes[0,2].set_xlabel('Mean Pixel Intensity')\n            axes[0,2].set_ylabel('Frequency')\n            axes[0,2].set_title('Mean Intensity Distribution')\n            axes[0,2].legend()\n        \n        # 4. Mask properties if available\n        if hasattr(self, 'mask_properties_df'):\n            axes[1,0].hist(self.mask_properties_df['forgery_ratio'], bins=20, alpha=0.7, color='purple')\n            axes[1,0].set_xlabel('Forgery Pixel Ratio')\n            axes[1,0].set_ylabel('Frequency')\n            axes[1,0].set_title('Forgery Area Distribution')\n            \n            # 5. Unique values in masks\n            axes[1,1].hist(self.mask_properties_df['unique_values'], bins=20, alpha=0.7, color='orange')\n            axes[1,1].set_xlabel('Number of Unique Values in Mask')\n            axes[1,1].set_ylabel('Frequency')\n            axes[1,1].set_title('Mask Complexity (Unique Values)')\n            \n            # 6. Mask format distribution\n            if not self.mask_properties_df.empty:\n                format_counts = self.mask_properties_df['mask_format'].value_counts()\n                axes[1,2].bar(format_counts.index.astype(str), format_counts.values, color='teal', alpha=0.7)\n                axes[1,2].set_xlabel('Mask Format (shape)')\n                axes[1,2].set_ylabel('Count')\n                axes[1,2].set_title('Mask Format Distribution')\n                axes[1,2].tick_params(axis='x', rotation=45)\n        \n        plt.tight_layout()\n        plt.show()\n    \n    def display_sample_images(self, n_samples=5):\n        \"\"\"Display sample images with their masks if available\"\"\"\n        print(\"\\n🖼️ SAMPLE IMAGES VISUALIZATION\")\n        print(\"=\" * 50)\n        \n        # Get sample images from each category\n        categories = {\n            'Authentic': self.train_authentic_dir,\n            'Forged': self.train_forged_dir,\n            'Test': self.test_images_dir\n        }\n        \n        for category, directory in categories.items():\n            if not directory.exists():\n                continue\n                \n            print(f\"\\n{category} Images:\")\n            image_files = list(directory.glob('*.png'))[:n_samples]\n            \n            if not image_files:\n                print(f\"No images found in {category}\")\n                continue\n            \n            fig, axes = plt.subplots(2, n_samples, figsize=(4*n_samples, 8))\n            if n_samples == 1:\n                axes = axes.reshape(2, 1)\n            \n            for idx, img_path in enumerate(image_files):\n                if idx >= n_samples:\n                    break\n                    \n                # Load and display original image\n                img = cv2.imread(str(img_path))\n                img_rgb = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n                \n                axes[0, idx].imshow(img_rgb)\n                axes[0, idx].set_title(f'{category}\\n{img_path.name}')\n                axes[0, idx].axis('off')\n                \n                # Try to load and display corresponding mask for forged images\n                if category == 'Forged':\n                    mask_path = self.train_masks_dir / f\"{img_path.stem}.npy\"\n                    if mask_path.exists():\n                        mask = decode_mask(mask_path, img.shape)\n                        axes[1, idx].imshow(mask, cmap='hot')\n                        axes[1, idx].set_title(f'Decoded Mask\\n{mask_path.name}')\n                    else:\n                        axes[1, idx].text(0.5, 0.5, 'Mask not found', \n                                         ha='center', va='center', transform=axes[1, idx].transAxes)\n                    axes[1, idx].axis('off')\n                else:\n                    axes[1, idx].text(0.5, 0.5, 'No mask', \n                                     ha='center', va='center', transform=axes[1, idx].transAxes)\n                    axes[1, idx].axis('off')\n            \n            plt.tight_layout()\n            plt.show()\n    \n    def display_overlay_images(self, n_samples=3):\n        \"\"\"Display images with mask overlays\"\"\"\n        print(\"\\n🎭 IMAGE-MASK OVERLAY VISUALIZATION\")\n        print(\"=\" * 50)\n        \n        if not self.train_forged_dir.exists():\n            print(\"Forged images directory not found!\")\n            return\n        \n        forged_images = list(self.train_forged_dir.glob('*.png'))[:n_samples]\n        \n        for img_path in forged_images:\n            # Load image\n            img = cv2.imread(str(img_path))\n            img_rgb = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n            \n            # Load and decode mask\n            mask_path = self.train_masks_dir / f\"{img_path.stem}.npy\"\n            if not mask_path.exists():\n                print(f\"Mask not found for {img_path.name}\")\n                continue\n            \n            mask = decode_mask(mask_path, img.shape)\n            original_mask_data = np.load(mask_path)\n            \n            # Create overlay\n            fig, axes = plt.subplots(2, 3, figsize=(18, 12))\n            \n            # Original image\n            axes[0, 0].imshow(img_rgb)\n            axes[0, 0].set_title(f'Original Image\\n{img_path.name}')\n            axes[0, 0].axis('off')\n            \n            # Original mask data (first channel if 3D)\n            if original_mask_data.ndim == 3:\n                axes[0, 1].imshow(original_mask_data[0] if original_mask_data.shape[0] >= 1 else original_mask_data)\n                axes[0, 1].set_title(f'Original Mask Channel 0\\nShape: {original_mask_data.shape}')\n            else:\n                axes[0, 1].imshow(original_mask_data)\n                axes[0, 1].set_title(f'Original Mask Data\\nShape: {original_mask_data.shape}')\n            axes[0, 1].axis('off')\n            \n            # Decoded binary mask\n            axes[0, 2].imshow(mask, cmap='hot')\n            axes[0, 2].set_title('Decoded Binary Mask')\n            axes[0, 2].axis('off')\n            \n            # Overlay 1: Simple overlay\n            axes[1, 0].imshow(img_rgb)\n            axes[1, 0].imshow(mask, cmap='hot', alpha=0.5)\n            axes[1, 0].set_title('Image with Mask Overlay (Alpha=0.5)')\n            axes[1, 0].axis('off')\n            \n            # Overlay 2: Contour overlay\n            axes[1, 1].imshow(img_rgb)\n            contours, _ = cv2.findContours(mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n            contour_img = img_rgb.copy()\n            cv2.drawContours(contour_img, contours, -1, (0, 255, 0), 2)\n            axes[1, 1].imshow(contour_img)\n            axes[1, 1].set_title('Image with Contour Overlay')\n            axes[1, 1].axis('off')\n            \n            # Forgery statistics\n            axes[1, 2].axis('off')\n            forgery_ratio = np.sum(mask > 0) / mask.size\n            stats_text = f\"Forgery Statistics:\\n\"\n            stats_text += f\"Coverage: {forgery_ratio:.2%}\\n\"\n            stats_text += f\"Forged Pixels: {np.sum(mask > 0):,}\\n\"\n            stats_text += f\"Total Pixels: {mask.size:,}\\n\"\n            stats_text += f\"Original Mask Shape: {original_mask_data.shape}\\n\"\n            stats_text += f\"Decoded Mask Shape: {mask.shape}\"\n            \n            axes[1, 2].text(0.1, 0.9, stats_text, transform=axes[1, 2].transAxes, \n                           fontsize=10, verticalalignment='top', bbox=dict(boxstyle='round', facecolor='wheat', alpha=0.5))\n            \n            plt.tight_layout()\n            plt.show()\n            \n            print(\"-\" * 50)\n    \n    def analyze_mask_formats(self):\n        \"\"\"Detailed analysis of different mask formats in the dataset\"\"\"\n        print(\"\\n🔍 DETAILED MASK FORMAT ANALYSIS\")\n        print(\"=\" * 50)\n        \n        if not self.train_masks_dir.exists():\n            print(\"Masks directory not found!\")\n            return\n        \n        mask_files = list(self.train_masks_dir.glob('*.npy'))\n        format_stats = {}\n        \n        for mask_path in tqdm(mask_files, desc=\"Analyzing mask formats\"):\n            try:\n                mask_data = np.load(mask_path)\n                shape_str = str(mask_data.shape)\n                \n                if shape_str not in format_stats:\n                    format_stats[shape_str] = {\n                        'count': 0,\n                        'shapes': set(),\n                        'dtypes': set(),\n                        'sample_files': []\n                    }\n                \n                format_stats[shape_str]['count'] += 1\n                format_stats[shape_str]['shapes'].add(mask_data.shape)\n                format_stats[shape_str]['dtypes'].add(str(mask_data.dtype))\n                if len(format_stats[shape_str]['sample_files']) < 3:\n                    format_stats[shape_str]['sample_files'].append(mask_path.name)\n                    \n            except Exception as e:\n                print(f\"Error analyzing {mask_path}: {e}\")\n        \n        print(\"\\nMask Format Summary:\")\n        for format_type, stats in format_stats.items():\n            print(f\"\\nFormat: {format_type}\")\n            print(f\"  Count: {stats['count']}\")\n            print(f\"  Dtypes: {list(stats['dtypes'])}\")\n            print(f\"  Sample files: {stats['sample_files']}\")\n    \n    def create_summary_report(self):\n        \"\"\"Generate a comprehensive EDA summary report\"\"\"\n        print(\"🚀 COMPREHENSIVE EDA SUMMARY REPORT\")\n        print(\"=\" * 60)\n        \n        # 1. Directory structure\n        self.explore_directory_structure()\n        \n        # 2. Basic statistics\n        stats = self.get_dataset_stats()\n        \n        # 3. Image properties\n        img_df = self.analyze_image_properties(sample_size=100)\n        \n        # 4. Mask properties\n        mask_df = self.analyze_mask_properties(sample_size=100)\n        \n        # 5. Mask format analysis\n        self.analyze_mask_formats()\n        \n        # 6. Visualizations\n        self.visualize_image_distributions()\n        self.display_sample_images(n_samples=3)\n        self.display_overlay_images(n_samples=2)\n        \n        # 7. Key insights\n        print(\"\\n💡 KEY INSIGHTS AND OBSERVATIONS\")\n        print(\"=\" * 40)\n        \n        if hasattr(self, 'img_properties_df'):\n            print(f\"• Image dimensions vary from {img_df['width'].min()}x{img_df['height'].min()} to {img_df['width'].max()}x{img_df['height'].max()}\")\n            print(f\"• Average image size: {img_df['width'].mean():.0f}x{img_df['height'].mean():.0f}\")\n            print(f\"• Common aspect ratios: {img_df['aspect_ratio'].value_counts().head(3).to_dict()}\")\n        \n        if hasattr(self, 'mask_properties_df') and not self.mask_properties_df.empty:\n            print(f\"• Average forgery coverage: {self.mask_properties_df['forgery_ratio'].mean():.2%}\")\n            print(f\"• Mask complexity (avg unique values): {self.mask_properties_df['unique_values'].mean():.1f}\")\n            print(f\"• Forgery size range: {self.mask_properties_df['forged_pixels'].min()} to {self.mask_properties_df['forged_pixels'].max()} pixels\")\n        \n\n# Usage example\ndef main():\n    # Initialize EDA pipeline\n    data_dir = \"/kaggle/input/recodai-luc-scientific-image-forgery-detection\"  # Update this path\n    eda = ScientificImageEDA(data_dir)\n    \n    # Run complete analysis\n    eda.create_summary_report()\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:25:09.729390Z","iopub.execute_input":"2025-10-24T09:25:09.729698Z","iopub.status.idle":"2025-10-24T09:26:11.157544Z","shell.execute_reply.started":"2025-10-24T09:25:09.729676Z","shell.execute_reply":"2025-10-24T09:26:11.156468Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}