{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Basic Libraries\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\n\n# Image Libraries\nimport cv2\nimport os\n\n# For displaying images\nfrom PIL import Image","metadata":{"execution":{"iopub.status.busy":"2024-09-25T18:10:09.42547Z","iopub.execute_input":"2024-09-25T18:10:09.426323Z","iopub.status.idle":"2024-09-25T18:10:09.432508Z","shell.execute_reply.started":"2024-09-25T18:10:09.426276Z","shell.execute_reply":"2024-09-25T18:10:09.431446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 1: Check the folder structure and contents\ninput_dir = \"/kaggle/input/histopathologic-cancer-detection\"\nprint(\"Contents of input directory:\", os.listdir(input_dir))\n\n# Step 2: Load the train labels CSV\nlabels_path = os.path.join(input_dir, \"train_labels.csv\")\nlabels = pd.read_csv(labels_path)\n\n# Display the first few rows of the labels CSV\nprint(labels.head())\n\n# Step 3: Check the class distribution (tumor vs. no tumor)\nprint(labels['label'].value_counts())\n\n# Plot class distribution\nsns.countplot(x='label', data=labels)\nplt.title('Class Distribution (Cancer vs. Non-Cancer)')\nplt.show()\n\n# Step 4: Display sample images from the train folder\ntrain_dir = os.path.join(input_dir, \"train\")\nimage_files = os.listdir(train_dir)[:10]  # Load a few images for visualization\n\n# Plot a few sample images\nplt.figure(figsize=(10, 10))\nfor i, file in enumerate(image_files):\n    image = Image.open(os.path.join(train_dir, file))\n    plt.subplot(4, 4, i + 1)\n    plt.imshow(image)\n    plt.title(f\"Image: {file}\")\n    plt.axis('off')\nplt.show()\n\n# Step 5: Check if train_labels.csv correctly maps to the images in the train folder\nprint(\"Example image filename:\", image_files[0])\nexample_image_id = image_files[0].replace(\".tif\", \"\")\nprint(\"Label for the example image:\", labels[labels['id'] == example_image_id]['label'].values[0])\n\n","metadata":{"execution":{"iopub.status.busy":"2024-09-25T18:10:15.212386Z","iopub.execute_input":"2024-09-25T18:10:15.212803Z","iopub.status.idle":"2024-09-25T18:10:38.139329Z","shell.execute_reply.started":"2024-09-25T18:10:15.212762Z","shell.execute_reply":"2024-09-25T18:10:38.138061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nfrom torch.utils.data import DataLoader, Dataset\nfrom torchvision import transforms\nimport random\n\n# Defining paths\ndata_dir = '/kaggle/input/histopathologic-cancer-detection/train/'\nlabels_path = '/kaggle/input/histopathologic-cancer-detection/train_labels.csv'\n\n# Loading the labels\nlabels_df = pd.read_csv(labels_path)\n\n# Setting a threshold for percentage of identical pixels \nthreshold_percentage = 0.95\n\n# Select device (GPU/CPU)\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(f\"Using device: {device}\")\n\n# Transform to convert images to tensors and apply CenterCrop\ntransform = transforms.Compose([\n    transforms.CenterCrop(32),  # Cropping the center to 32x32\n    transforms.ToTensor()       # Converting image to tensor\n])\n\n# Efficient Dataset class for loading images\nclass PCamDataset(Dataset):\n    def __init__(self, labels_df, data_dir, transform=None):\n        self.labels_df = labels_df\n        self.data_dir = data_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.labels_df)\n\n    def __getitem__(self, idx):\n        img_id = self.labels_df.iloc[idx]['id']\n        img_path = os.path.join(self.data_dir, img_id + '.tif')\n        img = Image.open(img_path).convert('RGB')\n        if self.transform:\n            img = self.transform(img)\n        return img_id, img\n\n# Optimized function to check noisy images (batch-wise)\ndef is_noisy_image_batch(batch_tensor, threshold_percentage):\n    \"\"\"\n    This function takes a batch of images and checks for noise in a vectorized manner.\n    \"\"\"\n    # Flatten the images into (batch_size, num_pixels, 3)\n    flattened_batch = batch_tensor.view(batch_tensor.size(0), -1, batch_tensor.size(1))\n\n    # Check for identical pixels per image\n    noisy_images = []\n    for img in flattened_batch:\n        unique_pixels, counts = img.unique(dim=0, return_counts=True)\n        max_pixel_percentage = counts.max().item() / counts.sum().item()\n        noisy_images.append(max_pixel_percentage >= threshold_percentage)\n\n    return torch.tensor(noisy_images, device=device)\n\n# Initialize dataset and dataloader\ndataset = PCamDataset(labels_df, data_dir, transform=transform)\ndataloader = DataLoader(dataset, batch_size=128, shuffle=False, num_workers=6, pin_memory=False, persistent_workers=True)\n\n# Lists to store valid (non-noisy) and noisy image IDs\nvalid_image_ids = []\nnoisy_image_ids = []\nnoisy_images_samples = []\n\n# Use a progress bar to monitor processing\nwith tqdm(total=len(labels_df), desc=\"Processing Images\", unit=\"image\") as pbar:\n    for img_ids, imgs in dataloader:\n        imgs = imgs.to(device)  # Move batch to GPU\n        is_noisy_batch = is_noisy_image_batch(imgs, threshold_percentage)\n\n        # Add non-noisy and noisy image IDs to their respective lists\n        valid_image_ids.extend([img_id for img_id, noisy in zip(img_ids, is_noisy_batch) if not noisy])\n        noisy_image_ids.extend([img_id for img_id, noisy in zip(img_ids, is_noisy_batch) if noisy])\n        \n        # Collecting noisy images for later display\n        noisy_images_samples.extend([img.cpu() for img, noisy in zip(imgs, is_noisy_batch) if noisy])\n\n        pbar.update(len(img_ids))  # Update progress bar\n\n# Filtering out the labels of valid (clean) and noisy images\nclean_labels_df = labels_df[labels_df['id'].isin(valid_image_ids)]\nnoisy_labels_df = labels_df[labels_df['id'].isin(noisy_image_ids)]\n\n# Saving the clean and noisy labels to CSV files\nclean_labels_df.to_csv('clean_train_labels.csv', index=False)\nnoisy_labels_df.to_csv('noisy_train_labels.csv', index=False)\n\n# Print the total valid and noisy images\nprint(f\"Total valid images: {len(valid_image_ids)}\")\nprint(f\"Total noisy images: {len(noisy_image_ids)}\")\n\n# Function to display some random noisy images\ndef display_noisy_images(noisy_images_samples, num_images=5):\n    plt.figure(figsize=(10, 10))\n    for i in range(num_images):\n        img_tensor = random.choice(noisy_images_samples)  # Randomly select a noisy image\n        img_np = img_tensor.permute(1, 2, 0).numpy()  # Convert tensor to numpy for displaying\n        plt.subplot(1, num_images, i + 1)\n        plt.imshow(img_np)\n        plt.axis('off')\n    plt.show()\n\n# Display 5 random noisy images\ndisplay_noisy_images(noisy_images_samples, num_images=5)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-25T19:32:37.424798Z","iopub.execute_input":"2024-09-25T19:32:37.426486Z","iopub.status.idle":"2024-09-25T19:48:20.460885Z","shell.execute_reply.started":"2024-09-25T19:32:37.426396Z","shell.execute_reply":"2024-09-25T19:48:20.458594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Custom Dataset class optimized for loading images\nclass CustomDataset(Dataset):\n    def __init__(self, labels_df, data_dir, transform=None):\n        self.labels_df = labels_df\n        self.data_dir = data_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.labels_df)\n\n    def __getitem__(self, idx):\n        img_id = self.labels_df.iloc[idx]['id']\n        label = self.labels_df.iloc[idx]['label']\n        img_path = os.path.join(self.data_dir, img_id + '.tif')\n\n        # Load image and apply transformations\n        image = Image.open(img_path).convert('RGB')\n        if self.transform:\n            image = self.transform(image)\n\n        return image, label\n\n# Transform to convert images to tensors and apply CenterCrop\ntransform = transforms.Compose([\n    transforms.CenterCrop(32),  # Cropping the center to 32x32\n    transforms.ToTensor()       # Converting image to tensor\n])\n\n\n# Create the dataset with transformations\ndataset = CustomDataset(labels_df, data_dir, transform=transform)\n\n# Define the batch size and DataLoader settings\nbatch_size = 64  # You can increase this based on your hardware (e.g., GPU memory)\nnum_workers = 4   # Adjust based on CPU cores for parallel data loading\n\n# Create DataLoader\ndataloader = DataLoader(dataset, batch_size=batch_size, shuffle=True, num_workers=num_workers, pin_memory=True)\n\n# Check batch shapes with tqdm progress bar\nwith tqdm(dataloader, desc=\"Loading images\", unit=\"batch\") as loader:\n    for images, labels in loader:\n        print(f\"Shape of single image: {images[0].shape}\")  # Shape of a single image\n        print(f\"Shape of full image batch: {images.shape}\")  # Shape of full batch of images\n        break  # Only load one batch for inspection\n","metadata":{"execution":{"iopub.status.busy":"2024-09-25T19:48:45.611293Z","iopub.execute_input":"2024-09-25T19:48:45.611784Z","iopub.status.idle":"2024-09-25T19:48:46.525115Z","shell.execute_reply.started":"2024-09-25T19:48:45.611743Z","shell.execute_reply":"2024-09-25T19:48:46.523704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to compute mean and std in a more efficient way\ndef calculate_mean_std(dataloader, device='cpu'):\n    # Initialize variables to store sums of means and squared means\n    total_sum = 0.0\n    total_sq_sum = 0.0\n    total_pixels = 0\n\n    # Move all computations to the appropriate device (GPU if available)\n    for images, _ in tqdm(dataloader, desc=\"Calculating mean and std\"):\n        images = images.to(device)\n\n        # Reshape the images into [batch_size, channels, height*width] and calculate the sum and squared sum\n        pixels_in_batch = images.size(0) * images.size(2) * images.size(3)\n        total_pixels += pixels_in_batch\n\n        total_sum += images.sum(dim=[0, 2, 3])\n        total_sq_sum += (images ** 2).sum(dim=[0, 2, 3])\n\n    # Calculate mean and std over all the pixels\n    mean = total_sum / total_pixels\n    std = torch.sqrt((total_sq_sum / total_pixels) - mean**2)\n\n    return mean.cpu(), std.cpu()  # Move results back to CPU if necessary\n\n# Assuming dataloader and device have been set up (e.g., device='cuda' if using a GPU)\nmean, std = calculate_mean_std(dataloader, device=device)\n\nprint(f'Mean: {mean}')\nprint(f'Std: {std}')\n","metadata":{"execution":{"iopub.status.busy":"2024-09-25T19:50:20.959802Z","iopub.execute_input":"2024-09-25T19:50:20.961477Z","iopub.status.idle":"2024-09-25T19:54:00.05218Z","shell.execute_reply.started":"2024-09-25T19:50:20.961388Z","shell.execute_reply":"2024-09-25T19:54:00.050751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom PIL import Image\nfrom sklearn.model_selection import train_test_split\nimport torch\nfrom torch.utils.data import DataLoader, Dataset\nfrom torchvision import transforms\n\n# Custom Dataset class for handling PCam data\nclass PCamDataset(Dataset):\n    def __init__(self, labels_df, data_dir, transform=None):\n        self.labels_df = labels_df\n        self.data_dir = data_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.labels_df)\n\n    def __getitem__(self, idx):\n        img_id = self.labels_df.iloc[idx]['id']\n        img_path = os.path.join(self.data_dir, img_id + '.tif')\n        img = Image.open(img_path).convert('RGB')\n        label = self.labels_df.iloc[idx]['label']  # Assuming 'label' column exists in the DataFrame\n        if self.transform:\n            img = self.transform(img)\n        return img, label\n\n# Load clean labels DataFrame\nclean_labels_df = pd.read_csv('/kaggle/working/clean_train_labels.csv')\n\n# Define the path to the image data\ndata_dir = '/kaggle/input/histopathologic-cancer-detection/train/'  # Update this path as needed\n\n# 70/10/20 Split: Split dataset into train, validation, and test sets\ntrain_df, temp_df = train_test_split(clean_labels_df, test_size=0.30, stratify=clean_labels_df['label'], random_state=42)\nval_df, test_df = train_test_split(temp_df, test_size=0.33, stratify=temp_df['label'], random_state=42)\n\n# Print the number of images in each split\nprint(f\"Train set size: {len(train_df)} images\")\nprint(f\"Validation set size: {len(val_df)} images\")\nprint(f\"Test set size: {len(test_df)} images\")\n\n# Define transforms (without data augmentation as requested)\nbasic_transform = transforms.Compose([\n    transforms.ToTensor()\n])\n\n# Create datasets\ntrain_dataset = PCamDataset(train_df, data_dir, transform=basic_transform)\nval_dataset = PCamDataset(val_df, data_dir, transform=basic_transform)\ntest_dataset = PCamDataset(test_df, data_dir, transform=basic_transform)\n\n# Define batch size\nbatch_size = 128\n\n# Create DataLoaders\ntrain_loader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True, num_workers=4)\nval_loader = DataLoader(val_dataset, batch_size=batch_size, shuffle=False, num_workers=4)\ntest_loader = DataLoader(test_dataset, batch_size=batch_size, shuffle=False, num_workers=4)\n\nprint(\"DataLoaders created for train, validation, and test sets.\")\n\n","metadata":{"execution":{"iopub.status.busy":"2024-09-25T20:27:17.465591Z","iopub.execute_input":"2024-09-25T20:27:17.466127Z","iopub.status.idle":"2024-09-25T20:27:17.914219Z","shell.execute_reply.started":"2024-09-25T20:27:17.466076Z","shell.execute_reply":"2024-09-25T20:27:17.912928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to calculate and display the percentage distribution of labels\ndef print_distribution(label_counts, dataset_name):\n    total = label_counts.sum()\n    non_cancerous_percentage = (label_counts[0] / total) * 100\n    cancerous_percentage = (label_counts[1] / total) * 100\n    \n    print(f\"{dataset_name} set distribution:\")\n    print(f\"Non-cancerous: {label_counts[0]} ({non_cancerous_percentage:.2f}%)\")\n    print(f\"Cancerous: {label_counts[1]} ({cancerous_percentage:.2f}%)\\n\")\n\n# Get the label counts for each dataset\ntrain_distribution = train_df['label'].value_counts()\nval_distribution = val_df['label'].value_counts()\ntest_distribution = test_df['label'].value_counts()\n\n# Print the percentage distribution for each dataset\nprint_distribution(train_distribution, \"Train\")\nprint_distribution(val_distribution, \"Validation\")\nprint_distribution(test_distribution, \"Test\")\n","metadata":{"execution":{"iopub.status.busy":"2024-09-25T20:30:37.469571Z","iopub.execute_input":"2024-09-25T20:30:37.470042Z","iopub.status.idle":"2024-09-25T20:30:37.483316Z","shell.execute_reply.started":"2024-09-25T20:30:37.469975Z","shell.execute_reply":"2024-09-25T20:30:37.481902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nfrom torch.utils.data import DataLoader, Subset\nimport random\nfrom tqdm import tqdm  # Import tqdm for progress bar\n\n# Function to create a balanced data loader with a specified ratio of cancerous to non-cancerous samples\ndef create_balanced_loader(data_loader, cancerous_label, non_cancerous_label, cancerous_ratio=0.2):\n    cancerous_indices = []\n    non_cancerous_indices = []\n    \n    # Separating cancerous and non-cancerous samples\n    print(\"Separating cancerous and non-cancerous samples...\")\n    for idx, (_, labels) in tqdm(enumerate(data_loader.dataset), total=len(data_loader.dataset), desc=\"Processing\"):\n        if labels == cancerous_label:\n            cancerous_indices.append(idx)\n        else:\n            non_cancerous_indices.append(idx)\n    \n    # Shuffle and select indices\n    random.shuffle(cancerous_indices)\n    random.shuffle(non_cancerous_indices)\n    \n    num_cancerous = int(len(cancerous_indices) * cancerous_ratio)\n    num_non_cancerous = int(len(non_cancerous_indices) * (1 - cancerous_ratio))\n    \n    selected_cancerous_indices = cancerous_indices[:num_cancerous]\n    selected_non_cancerous_indices = non_cancerous_indices[:num_non_cancerous]\n    \n    # Combine and shuffle the selected indices\n    new_indices = selected_cancerous_indices + selected_non_cancerous_indices\n    random.shuffle(new_indices)\n    \n    # Create the new dataset and data loader\n    new_subset = Subset(data_loader.dataset, new_indices)\n    new_loader = DataLoader(new_subset, batch_size=data_loader.batch_size, shuffle=True, drop_last=True)\n    \n    return new_loader\n\n# Function to check the distribution of samples in a data loader\ndef check_distribution(data_loader, cancerous_label):\n    cancer_count = 0\n    non_cancer_count = 0\n    \n    print(\"Checking distribution in the new data loader...\")\n    for _, labels in tqdm(data_loader, desc=\"Checking\"):\n        cancer_count += (labels == cancerous_label).sum().item()\n        non_cancer_count += (labels != cancerous_label).sum().item()\n    \n    total_count = cancer_count + non_cancer_count\n    return cancer_count, non_cancer_count, total_count\n\n# Define your cancerous and non-cancerous labels\ncancerous_label = 1\nnon_cancerous_label = 0\n\n# Create new data loaders with a balanced distribution for validation and test sets\nnew_val_loader = create_balanced_loader(val_loader, cancerous_label, non_cancerous_label, cancerous_ratio=0.2)\nnew_test_loader = create_balanced_loader(test_loader, cancerous_label, non_cancerous_label, cancerous_ratio=0.2)\n\n# Check and print the distribution for the new validation loader\nval_cancer_count, val_non_cancer_count, val_total_count = check_distribution(new_val_loader, cancerous_label)\nprint(f\"New Validation Loader Distribution: Cancerous: {val_cancer_count}, Non-Cancerous: {val_non_cancer_count}, Total: {val_total_count}\")\n\n# Check and print the distribution for the new test loader\ntest_cancer_count, test_non_cancer_count, test_total_count = check_distribution(new_test_loader, cancerous_label)\nprint(f\"New Test Loader Distribution: Cancerous: {test_cancer_count}, Non-Cancerous: {test_non_cancer_count}, Total: {test_total_count}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-09-25T20:44:47.895312Z","iopub.execute_input":"2024-09-25T20:44:47.895777Z","iopub.status.idle":"2024-09-25T20:49:47.736648Z","shell.execute_reply.started":"2024-09-25T20:44:47.895733Z","shell.execute_reply":"2024-09-25T20:49:47.73529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}