{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":1079953,"sourceType":"datasetVersion","datasetId":601280},{"sourceId":4431730,"sourceType":"datasetVersion","datasetId":2595427},{"sourceId":12745533,"sourceType":"datasetVersion","datasetId":672377}],"dockerImageVersionId":31089,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# List of dataset paths to analyze\ndataset_paths = [\n    \"/kaggle/input/brain-tumor-classification-mri\",\n    \"/kaggle/input/lung-and-colon-cancer-histopathological-images/lung_colon_image_set\",\n    \"/kaggle/input/skin-cancer-dataset\"\n]\n\n# Collect image information\ndata_info = []\n\ndef count_images_in_directory(directory_path):\n    \"\"\"Count image files in a directory (non-recursive)\"\"\"\n    if not os.path.exists(directory_path):\n        return 0\n    \n    return len([f for f in os.listdir(directory_path) \n               if f.lower().endswith(('.png', '.jpg', '.jpeg', '.bmp', '.tiff'))])\n\nfor dataset_path in dataset_paths:\n    dataset_name = os.path.basename(dataset_path)\n    \n    if not os.path.exists(dataset_path):\n        print(f\"⚠️ Path not found: {dataset_path}\")\n        continue\n    \n    print(f\"🔍 Scanning dataset: {dataset_name}\")\n    \n    # Check if the dataset has split folders or direct class folders\n    items = os.listdir(dataset_path)\n    subdirs = [d for d in items if os.path.isdir(os.path.join(dataset_path, d))]\n    \n    # Common split folder names\n    split_names = [\"train\", \"training\", \"test\", \"testing\", \"val\", \"validation\"]\n    \n    # Check if any subdirectory matches split names\n    has_splits = any(name.lower() in split_names for name in subdirs)\n    \n    if has_splits:\n        # Process datasets with split folders\n        for split in subdirs:\n            split_path = os.path.join(dataset_path, split)\n            if os.path.isdir(split_path):\n                for class_name in os.listdir(split_path):\n                    class_path = os.path.join(split_path, class_name)\n                    if os.path.isdir(class_path):\n                        images = count_images_in_directory(class_path)\n                        data_info.append({\n                            \"Dataset\": dataset_name,\n                            \"Split\": split,\n                            \"Class\": class_name,\n                            \"Number of Images\": images\n                        })\n    else:\n        # Process datasets with direct class folders\n        # For these datasets, we need to check if the subdirectories contain images or more subdirectories\n        for item in subdirs:\n            item_path = os.path.join(dataset_path, item)\n            \n            # Check if this directory contains images directly\n            direct_images = count_images_in_directory(item_path)\n            if direct_images > 0:\n                data_info.append({\n                    \"Dataset\": dataset_name,\n                    \"Split\": \"N/A\",\n                    \"Class\": item,\n                    \"Number of Images\": direct_images\n                })\n            else:\n                # This directory might contain subdirectories with images\n                for class_name in os.listdir(item_path):\n                    class_path = os.path.join(item_path, class_name)\n                    if os.path.isdir(class_path):\n                        images = count_images_in_directory(class_path)\n                        data_info.append({\n                            \"Dataset\": dataset_name,\n                            \"Split\": \"N/A\",\n                            \"Class\": class_name,\n                            \"Number of Images\": images\n                        })\n\n# Create DataFrame\nif data_info:\n    df = pd.DataFrame(data_info)\n    \n    # Display results\n    print(\"📊 Dataset Analysis Results:\")\n    print(\"=\" * 50)\n    print(df.to_string(index=False))\n    print(\"\\n\" + \"=\" * 50)\n    print(f\"Total Images Across All Datasets: {df['Number of Images'].sum()}\")\nelse:\n    print(\"❌ No image data found in any of the specified paths\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-22T18:43:20.804828Z","iopub.execute_input":"2025-08-22T18:43:20.805146Z","iopub.status.idle":"2025-08-22T18:43:20.860866Z","shell.execute_reply.started":"2025-08-22T18:43:20.805123Z","shell.execute_reply":"2025-08-22T18:43:20.859964Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate percentages\nif 'df' in locals() and not df.empty:\n    # Create a copy of the dataframe to avoid modifying the original\n    percentage_df = df.copy()\n    \n    # Calculate percentages only for datasets with images\n    def calculate_percentage(group):\n        total = group['Number of Images'].sum()\n        if total > 0:\n            return round((group['Number of Images'] / total) * 100, 2)\n        else:\n            return 0  # Return 0 if no images to avoid division by zero\n    \n    percentage_df['Percentage (%)'] = percentage_df.groupby('Dataset').apply(calculate_percentage).reset_index(level=0, drop=True)\n    \n    # Display results with percentages\n    print(\"📊 Dataset Analysis with Percentages:\")\n    print(\"=\" * 60)\n    print(percentage_df.to_string(index=False))\n    \n    # Display summary by dataset\n    print(\"\\n📈 Summary by Dataset:\")\n    print(\"=\" * 40)\n    summary = df.groupby('Dataset')['Number of Images'].agg(['sum', 'count'])\n    summary.columns = ['Total Images', 'Number of Classes']\n    print(summary.to_string())\nelse:\n    print(\"❌ No data available for percentage calculation\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T18:44:05.000724Z","iopub.execute_input":"2025-08-22T18:44:05.001070Z","iopub.status.idle":"2025-08-22T18:44:05.021050Z","shell.execute_reply.started":"2025-08-22T18:44:05.001046Z","shell.execute_reply":"2025-08-22T18:44:05.019978Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# 📊 Global class distribution (all datasets together)\nglobal_distribution = (\n    percentage_df.groupby(\"Class\")[\"Number of Images\"].sum().reset_index()\n)\n\n# skip if there are no images at all\nif global_distribution[\"Number of Images\"].sum() == 0:\n    print(\"⚠️ No images found in any dataset!\")\nelse:\n    plt.figure(figsize=(7, 7))\n    plt.pie(\n        global_distribution[\"Number of Images\"],\n        labels=global_distribution[\"Class\"],\n        autopct=\"%1.1f%%\",\n        startangle=90,\n        counterclock=False\n    )\n    plt.title(\"Global Class Distribution (All Datasets Combined)\")\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T18:44:09.173313Z","iopub.execute_input":"2025-08-22T18:44:09.173638Z","iopub.status.idle":"2025-08-22T18:44:09.334677Z","shell.execute_reply.started":"2025-08-22T18:44:09.173615Z","shell.execute_reply":"2025-08-22T18:44:09.333743Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\nif 'df' in locals() and not df.empty:\n    # Visualization 1: Bar chart by dataset and class\n    plt.figure(figsize=(14, 8))\n    \n    # Get unique datasets\n    datasets = df['Dataset'].unique()\n    \n    # Create a color map for datasets\n    colors = plt.cm.tab10(np.linspace(0, 1, len(datasets)))\n    \n    # Create positions for bars\n    x_pos = np.arange(len(df))\n    \n    # Plot bars\n    for i, dataset in enumerate(datasets):\n        subset = df[df['Dataset'] == dataset]\n        indices = df[df['Dataset'] == dataset].index\n        plt.bar(indices, subset['Number of Images'], color=colors[i], label=dataset, alpha=0.8)\n    \n    plt.xlabel('Classes')\n    plt.ylabel('Number of Images')\n    plt.title('Image Distribution Across Datasets and Classes')\n    plt.xticks(x_pos, df['Class'], rotation=45, ha='right')\n    plt.legend()\n    plt.tight_layout()\n    plt.show()\n    \n    # Visualization 2: Pie charts for each dataset\n    for dataset in df['Dataset'].unique():\n        subset = df[df['Dataset'] == dataset]\n        if subset['Number of Images'].sum() > 0:\n            plt.figure(figsize=(8, 8))\n            plt.pie(subset['Number of Images'], \n                    labels=subset['Class'], \n                    autopct='%1.1f%%',\n                    startangle=90)\n            plt.title(f'Class Distribution in {dataset}')\n            plt.axis('equal')\n            plt.tight_layout()\n            plt.show()\n        else:\n            print(f\"⚠️ Skipping {dataset} (no images found)\")\n    \n    # Visualization 3: Global distribution across all datasets\n    global_dist = df.groupby('Class')['Number of Images'].sum()\n    if global_dist.sum() > 0:\n        plt.figure(figsize=(10, 10))\n        plt.pie(global_dist.values, \n                labels=global_dist.index, \n                autopct='%1.1f%%',\n                startangle=90)\n        plt.title('Global Class Distribution (All Datasets Combined)')\n        plt.axis('equal')\n        plt.tight_layout()\n        plt.show()\n    else:\n        print(\"⚠️ No images found in any dataset for global distribution\")\nelse:\n    print(\"❌ No data available for visualizations\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T18:44:12.932125Z","iopub.execute_input":"2025-08-22T18:44:12.932403Z","iopub.status.idle":"2025-08-22T18:44:13.895533Z","shell.execute_reply.started":"2025-08-22T18:44:12.932386Z","shell.execute_reply":"2025-08-22T18:44:13.894520Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport hashlib\nfrom PIL import Image\nimport numpy as np\n\ndef calculate_image_hash(image_path, hash_size=8):\n    \"\"\"\n    Calculate perceptual hash of an image\n    \"\"\"\n    try:\n        # Open and resize image\n        image = Image.open(image_path)\n        image = image.convert(\"L\").resize((hash_size+1, hash_size), Image.LANCZOS)\n        \n        # Calculate difference hash\n        pixels = list(image.getdata())\n        diff = [pixels[i] > pixels[i+1] for i in range(len(pixels)-1)]\n        \n        # Convert to hexadecimal hash\n        decimal_value = 0\n        hex_hash = []\n        for i, value in enumerate(diff):\n            if value:\n                decimal_value += 2**(i % 8)\n            if (i % 8) == 7:\n                hex_hash.append(hex(decimal_value)[2:].rjust(2, '0'))\n                decimal_value = 0\n        \n        return ''.join(hex_hash)\n    except Exception as e:\n        print(f\"Error processing {image_path}: {e}\")\n        return None\n\ndef find_duplicate_images(dataset_paths):\n    \"\"\"\n    Find duplicate images across multiple datasets\n    \"\"\"\n    image_hashes = {}\n    duplicates = []\n    \n    for dataset_path in dataset_paths:\n        dataset_name = os.path.basename(dataset_path)\n        \n        if not os.path.exists(dataset_path):\n            print(f\"⚠️ Path not found: {dataset_path}\")\n            continue\n        \n        print(f\"🔍 Scanning for duplicates in: {dataset_name}\")\n        \n        # Walk through all directories\n        for root, dirs, files in os.walk(dataset_path):\n            for file in files:\n                if file.lower().endswith(('.png', '.jpg', '.jpeg', '.bmp', '.tiff')):\n                    image_path = os.path.join(root, file)\n                    \n                    # Calculate image hash\n                    image_hash = calculate_image_hash(image_path)\n                    \n                    if image_hash:\n                        # Get relative path for reporting\n                        rel_path = os.path.relpath(image_path, dataset_path)\n                        \n                        if image_hash in image_hashes:\n                            # Found a duplicate\n                            duplicates.append({\n                                \"hash\": image_hash,\n                                \"original_path\": image_hashes[image_hash][\"path\"],\n                                \"original_dataset\": image_hashes[image_hash][\"dataset\"],\n                                \"duplicate_path\": image_path,\n                                \"duplicate_dataset\": dataset_name,\n                                \"relative_path\": rel_path\n                            })\n                        else:\n                            # First time seeing this image\n                            image_hashes[image_hash] = {\n                                \"path\": image_path,\n                                \"dataset\": dataset_name\n                            }\n    \n    return duplicates\n\n# List of dataset paths to check for duplicates (without X-ray dataset)\ndataset_paths = [\n    \"/kaggle/input/brain-tumor-classification-mri\",\n    \"/kaggle/input/lung-and-colon-cancer-histopathological-images/lung_colon_image_set\",\n    \"/kaggle/input/skin-cancer-dataset\"\n]\n\n# Find duplicates\nduplicates = find_duplicate_images(dataset_paths)\n\n# Display results\nif duplicates:\n    print(f\"\\n🔍 Found {len(duplicates)} potential duplicate images:\")\n    print(\"=\" * 80)\n    \n    # Create a DataFrame for better display\n    dup_df = pd.DataFrame(duplicates)\n    \n    # Group by hash to see all duplicates of the same image\n    grouped_duplicates = dup_df.groupby('hash').agg({\n        'original_path': 'first',\n        'original_dataset': 'first',\n        'duplicate_path': list,\n        'duplicate_dataset': list,\n        'relative_path': list\n    }).reset_index()\n    \n    print(f\"Found {len(grouped_duplicates)} unique images with duplicates\")\n    \n    # Display the first few duplicates\n    for i, (_, row) in enumerate(grouped_duplicates.head().iterrows()):\n        print(f\"\\nDuplicate Set {i+1}:\")\n        print(f\"Original: {row['original_path']} (Dataset: {row['original_dataset']})\")\n        print(f\"Duplicates: {len(row['duplicate_path'])} found\")\n        for j, dup_path in enumerate(row['duplicate_path']):\n            print(f\"  {j+1}. {dup_path} (Dataset: {row['duplicate_dataset'][j]})\")\n    \n    # Save detailed report to CSV\n    dup_df.to_csv(\"duplicate_images_report.csv\", index=False)\n    print(f\"\\n📊 Detailed report saved to 'duplicate_images_report.csv'\")\nelse:\n    print(\"\\n✅ No duplicate images found across all datasets!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T18:44:26.019347Z","iopub.execute_input":"2025-08-22T18:44:26.020210Z","iopub.status.idle":"2025-08-22T18:49:28.648254Z","shell.execute_reply.started":"2025-08-22T18:44:26.020182Z","shell.execute_reply":"2025-08-22T18:49:28.647417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\nimport pandas as pd\n\n# Create a function to copy datasets to a writable directory\ndef copy_datasets_to_working(input_paths, output_base=\"/kaggle/working/datasets\"):\n    \"\"\"\n    Copy datasets from read-only input to writable working directory\n    \"\"\"\n    # Create output directory\n    os.makedirs(output_base, exist_ok=True)\n    \n    copied_paths = []\n    \n    for input_path in input_paths:\n        dataset_name = os.path.basename(input_path)\n        output_path = os.path.join(output_base, dataset_name)\n        \n        print(f\"📂 Copying {dataset_name} to {output_path}\")\n        \n        try:\n            # Remove existing directory if it exists\n            if os.path.exists(output_path):\n                shutil.rmtree(output_path)\n                \n            # Copy the dataset\n            shutil.copytree(input_path, output_path)\n            copied_paths.append(output_path)\n            print(f\"✅ Successfully copied {dataset_name}\")\n        except Exception as e:\n            print(f\"❌ Error copying {dataset_name}: {e}\")\n    \n    return copied_paths\n\n# List of dataset paths to copy (without X-ray dataset)\ndataset_paths = [\n    \"/kaggle/input/brain-tumor-classification-mri\",\n    \"/kaggle/input/lung-and-colon-cancer-histopathological-images/lung_colon_image_set\",\n    \"/kaggle/input/skin-cancer-dataset\"\n]\n\n# Copy datasets to working directory\nwritable_datasets = copy_datasets_to_working(dataset_paths)\nprint(f\"📋 Writable datasets: {writable_datasets}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T18:54:17.944743Z","iopub.execute_input":"2025-08-22T18:54:17.945053Z","iopub.status.idle":"2025-08-22T18:55:23.183461Z","shell.execute_reply.started":"2025-08-22T18:54:17.945032Z","shell.execute_reply":"2025-08-22T18:55:23.182520Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Now run the duplicate detection on the writable datasets\n# (Use the duplicate detection code from earlier, but with the writable_datasets paths)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T18:56:03.684880Z","iopub.execute_input":"2025-08-22T18:56:03.685236Z","iopub.status.idle":"2025-08-22T18:56:03.689650Z","shell.execute_reply.started":"2025-08-22T18:56:03.685206Z","shell.execute_reply":"2025-08-22T18:56:03.688870Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport hashlib\nfrom PIL import Image\nimport numpy as np\n\ndef calculate_image_hash(image_path, hash_size=8):\n    \"\"\"\n    Calculate perceptual hash of an image\n    \"\"\"\n    try:\n        # Open and resize image\n        image = Image.open(image_path)\n        image = image.convert(\"L\").resize((hash_size+1, hash_size), Image.LANCZOS)\n        \n        # Calculate difference hash\n        pixels = list(image.getdata())\n        diff = [pixels[i] > pixels[i+1] for i in range(len(pixels)-1)]\n        \n        # Convert to hexadecimal hash\n        decimal_value = 0\n        hex_hash = []\n        for i, value in enumerate(diff):\n            if value:\n                decimal_value += 2**(i % 8)\n            if (i % 8) == 7:\n                hex_hash.append(hex(decimal_value)[2:].rjust(2, '0'))\n                decimal_value = 0\n        \n        return ''.join(hex_hash)\n    except Exception as e:\n        print(f\"Error processing {image_path}: {e}\")\n        return None\n\ndef find_and_remove_duplicates(dataset_paths):\n    \"\"\"\n    Find and remove duplicate images across multiple datasets\n    \"\"\"\n    image_hashes = {}\n    duplicates_found = 0\n    \n    for dataset_path in dataset_paths:\n        dataset_name = os.path.basename(dataset_path)\n        \n        if not os.path.exists(dataset_path):\n            print(f\"⚠️ Path not found: {dataset_path}\")\n            continue\n        \n        print(f\"🔍 Scanning for duplicates in: {dataset_name}\")\n        \n        # Walk through all directories\n        for root, dirs, files in os.walk(dataset_path):\n            for file in files:\n                if file.lower().endswith(('.png', '.jpg', '.jpeg', '.bmp', '.tiff')):\n                    image_path = os.path.join(root, file)\n                    \n                    # Calculate image hash\n                    image_hash = calculate_image_hash(image_path)\n                    \n                    if image_hash:\n                        if image_hash in image_hashes:\n                            # Found a duplicate - remove it\n                            try:\n                                os.remove(image_path)\n                                duplicates_found += 1\n                                print(f\"🗑️ Removed duplicate: {image_path}\")\n                            except Exception as e:\n                                print(f\"❌ Error removing {image_path}: {e}\")\n                        else:\n                            # First time seeing this image\n                            image_hashes[image_hash] = image_path\n    \n    return duplicates_found\n\n# List of dataset paths in the writable directory\nwritable_dataset_paths = [\n    \"/kaggle/working/datasets/brain-tumor-classification-mri\",\n    \"/kaggle/working/datasets/lung_colon_image_set\",\n    \"/kaggle/working/datasets/skin-cancer-dataset\"\n]\n\n# Find and remove duplicates\nduplicates_removed = find_and_remove_duplicates(writable_dataset_paths)\n\nprint(f\"\\n📊 Removal Summary:\")\nprint(f\"✅ Successfully removed {duplicates_removed} duplicate images\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T18:57:59.607830Z","iopub.execute_input":"2025-08-22T18:57:59.608162Z","iopub.status.idle":"2025-08-22T19:02:24.803702Z","shell.execute_reply.started":"2025-08-22T18:57:59.608139Z","shell.execute_reply":"2025-08-22T19:02:24.802818Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\ndef count_images_in_directory(directory_path):\n    \"\"\"Count all image files in a directory (recursively)\"\"\"\n    image_count = 0\n    image_extensions = ('.png', '.jpg', '.jpeg', '.bmp', '.tiff')\n    \n    for root, dirs, files in os.walk(directory_path):\n        for file in files:\n            if file.lower().endswith(image_extensions):\n                image_count += 1\n                \n    return image_count\n\ndef analyze_datasets(dataset_paths):\n    \"\"\"Analyze image distribution across multiple datasets\"\"\"\n    results = []\n    total_images = 0\n    \n    for dataset_path in dataset_paths:\n        dataset_name = os.path.basename(dataset_path)\n        \n        if not os.path.exists(dataset_path):\n            print(f\"⚠️ Path not found: {dataset_path}\")\n            continue\n        \n        print(f\"🔍 Analyzing: {dataset_name}\")\n        \n        # Count images in the entire dataset\n        dataset_total = count_images_in_directory(dataset_path)\n        total_images += dataset_total\n        \n        # Get breakdown by class/subdirectory\n        for root, dirs, files in os.walk(dataset_path):\n            # If we're at a directory that contains images directly\n            image_files = [f for f in files if f.lower().endswith(('.png', '.jpg', '.jpeg', '.bmp', '.tiff'))]\n            \n            if image_files:\n                class_name = os.path.basename(root)\n                parent_dir = os.path.basename(os.path.dirname(root))\n                \n                # For better classification, use parent directory name if it looks like a class\n                if parent_dir in ['Training', 'Testing', 'train', 'test']:\n                    class_label = f\"{parent_dir}/{class_name}\"\n                else:\n                    class_label = class_name\n                \n                results.append({\n                    \"Dataset\": dataset_name,\n                    \"Class\": class_label,\n                    \"Image Count\": len(image_files)\n                })\n    \n    return results, total_images\n\n# List of dataset paths to analyze\ndataset_paths = [\n    \"/kaggle/working/datasets/brain-tumor-classification-mri\",\n    \"/kaggle/working/datasets/lung_colon_image_set\", \n    \"/kaggle/working/datasets/skin-cancer-dataset\"\n]\n\n# Analyze the datasets\nresults, total_images = analyze_datasets(dataset_paths)\n\n# Create a DataFrame for better visualization\ndf = pd.DataFrame(results)\n\n# Display results\nprint(\"\\n📊 Image Distribution Across Datasets:\")\nprint(\"=\" * 50)\nprint(df.to_string(index=False))\n\nprint(f\"\\n📈 Total Images Across All Datasets: {total_images}\")\n\n# Summary by dataset\nprint(\"\\n📋 Summary by Dataset:\")\nprint(\"=\" * 30)\ndataset_summary = df.groupby('Dataset')['Image Count'].sum()\nfor dataset, count in dataset_summary.items():\n    print(f\"{dataset}: {count} images\")\n\n# Summary by class (top 10)\nprint(\"\\n🏷️ Top 10 Classes by Image Count:\")\nprint(\"=\" * 40)\nclass_summary = df.groupby('Class')['Image Count'].sum().sort_values(ascending=False)\nprint(class_summary.head(10).to_string())\n\n# Save results to CSV\ndf.to_csv(\"image_count_report.csv\", index=False)\nprint(f\"\\n💾 Report saved to 'image_count_report.csv'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T19:07:12.557863Z","iopub.execute_input":"2025-08-22T19:07:12.558203Z","iopub.status.idle":"2025-08-22T19:07:12.657671Z","shell.execute_reply.started":"2025-08-22T19:07:12.558181Z","shell.execute_reply":"2025-08-22T19:07:12.656788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Visualization: Image distribution by dataset\nplt.figure(figsize=(10, 6))\ndataset_summary.plot(kind='bar', color='skyblue')\nplt.title('Image Distribution by Dataset')\nplt.xlabel('Dataset')\nplt.ylabel('Number of Images')\nplt.xticks(rotation=45)\nplt.tight_layout()\nplt.show()\n\n# Visualization: Top classes\nplt.figure(figsize=(12, 6))\nclass_summary.head(10).plot(kind='bar', color='lightcoral')\nplt.title('Top 10 Classes by Image Count')\nplt.xlabel('Class')\nplt.ylabel('Number of Images')\nplt.xticks(rotation=45)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T19:09:00.134985Z","iopub.execute_input":"2025-08-22T19:09:00.135618Z","iopub.status.idle":"2025-08-22T19:09:00.620854Z","shell.execute_reply.started":"2025-08-22T19:09:00.135583Z","shell.execute_reply":"2025-08-22T19:09:00.619995Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\ndef create_image_metadata_csv(dataset_paths, output_csv=\"image_metadata.csv\"):\n    \"\"\"\n    Create a CSV file with image path, label, split, and source information\n    \"\"\"\n    image_data = []\n    \n    for dataset_path in dataset_paths:\n        dataset_name = os.path.basename(dataset_path)\n        \n        if not os.path.exists(dataset_path):\n            print(f\"⚠️ Path not found: {dataset_path}\")\n            continue\n            \n        print(f\"🔍 Processing: {dataset_name}\")\n        \n        # Walk through all directories\n        for root, dirs, files in os.walk(dataset_path):\n            for file in files:\n                if file.lower().endswith(('.png', '.jpg', '.jpeg', '.bmp', '.tiff')):\n                    image_path = os.path.join(root, file)\n                    \n                    # Extract label from directory structure\n                    label = os.path.basename(root)\n                    \n                    # Extract split information if available\n                    split = \"N/A\"\n                    parent_dir = os.path.basename(os.path.dirname(root))\n                    if parent_dir.lower() in ['training', 'test', 'train', 'testing']:\n                        split = parent_dir\n                    \n                    image_data.append({\n                        \"image_path\": image_path,\n                        \"label\": label,\n                        \"split\": split,\n                        \"source\": dataset_name\n                    })\n    \n    # Create DataFrame\n    df = pd.DataFrame(image_data)\n    \n    # Save to CSV\n    df.to_csv(output_csv, index=False)\n    print(f\"✅ CSV file created: {output_csv}\")\n    print(f\"📊 Total images recorded: {len(df)}\")\n    \n    return df\n\n# List of dataset paths\ndataset_paths = [\n    \"/kaggle/working/datasets/brain-tumor-classification-mri\",\n    \"/kaggle/working/datasets/lung_colon_image_set\",\n    \"/kaggle/working/datasets/skin-cancer-dataset\"\n]\n\n# Create the CSV file\ndf = create_image_metadata_csv(dataset_paths)\n\n# Show a sample of the data\nprint(\"\\n📋 Sample of the CSV data:\")\nprint(df.head(10).to_string(index=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T19:23:58.582545Z","iopub.execute_input":"2025-08-22T19:23:58.583466Z","iopub.status.idle":"2025-08-22T19:23:58.946164Z","shell.execute_reply.started":"2025-08-22T19:23:58.583434Z","shell.execute_reply":"2025-08-22T19:23:58.945317Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}