{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Basic Libraries\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\n\n# Deep Learning Libraries\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Flatten, Dropout\nfrom tensorflow.keras.optimizers import Adam\n\n# Image Libraries\nimport cv2\nimport os\n\n# For displaying images\nfrom PIL import Image\n","metadata":{"execution":{"iopub.status.busy":"2024-09-24T23:53:33.747012Z","iopub.execute_input":"2024-09-24T23:53:33.74823Z","iopub.status.idle":"2024-09-24T23:53:33.754335Z","shell.execute_reply.started":"2024-09-24T23:53:33.748189Z","shell.execute_reply":"2024-09-24T23:53:33.753358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 1: Check the folder structure and contents\ninput_dir = \"/kaggle/input/histopathologic-cancer-detection\"\nprint(\"Contents of input directory:\", os.listdir(input_dir))\n\n# Step 2: Load the train labels CSV\nlabels_path = os.path.join(input_dir, \"train_labels.csv\")\nlabels = pd.read_csv(labels_path)\n\n# Display the first few rows of the labels CSV\nprint(labels.head())\n\n# Step 3: Check the class distribution (tumor vs. no tumor)\nprint(labels['label'].value_counts())\n\n# Plot class distribution\nsns.countplot(x='label', data=labels)\nplt.title('Class Distribution (Cancer vs. Non-Cancer)')\nplt.show()\n\n# Step 4: Display sample images from the train folder\ntrain_dir = os.path.join(input_dir, \"train\")\nimage_files = os.listdir(train_dir)[:10]  # Load a few images for visualization\n\n# Plot a few sample images\nplt.figure(figsize=(10, 10))\nfor i, file in enumerate(image_files):\n    image = Image.open(os.path.join(train_dir, file))\n    plt.subplot(4, 4, i + 1)\n    plt.imshow(image)\n    plt.title(f\"Image: {file}\")\n    plt.axis('off')\nplt.show()\n\n# Step 5: Check if train_labels.csv correctly maps to the images in the train folder\nprint(\"Example image filename:\", image_files[0])\nexample_image_id = image_files[0].replace(\".tif\", \"\")\nprint(\"Label for the example image:\", labels[labels['id'] == example_image_id]['label'].values[0])\n\n","metadata":{"execution":{"iopub.status.busy":"2024-09-24T23:53:37.048312Z","iopub.execute_input":"2024-09-24T23:53:37.048679Z","iopub.status.idle":"2024-09-24T23:53:41.00432Z","shell.execute_reply.started":"2024-09-24T23:53:37.048647Z","shell.execute_reply":"2024-09-24T23:53:41.003337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nfrom torch.utils.data import DataLoader, Dataset\nfrom tqdm import tqdm\nfrom torchvision import transforms\n\n# Defining the paths\ndata_dir = '/kaggle/input/histopathologic-cancer-detection/train/'\nlabels_path = '/kaggle/input/histopathologic-cancer-detection/train_labels.csv'\n\n# Loading the labels\nlabels_df = pd.read_csv(labels_path)\n\n# Setting a threshold for percentage of identical pixels \nthreshold_percentage = 0.95\n\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(f\"Using device: {device}\")\n\n# Using Transform to convert PIL image to PyTorch tensor\ntransform = transforms.ToTensor()\n\n# Class which loads images efficiently\nclass PCamDataset(Dataset):\n    def __init__(self, labels_df, data_dir, transform=None):\n        self.labels_df = labels_df\n        self.data_dir = data_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.labels_df)\n\n    def __getitem__(self, idx):\n        img_id = self.labels_df.iloc[idx]['id']\n        img_path = os.path.join(self.data_dir, img_id + '.tif')\n        img = Image.open(img_path).convert('RGB')\n        if self.transform:\n            img = self.transform(img)\n        return img_id, img\n\n# Function to check if the image has a majority of identical pixels without sampling\ndef is_noisy_image(image_tensor, threshold_percentage):\n    img_flattened = image_tensor.view(-1, image_tensor.shape[-1])\n    \n    unique_pixels, counts = img_flattened.unique(dim=0, return_counts=True)\n    \n    # Find the percentage of the most common pixel\n    max_pixel_percentage = counts.max().item() / counts.sum().item()\n    \n    return max_pixel_percentage >= threshold_percentage\n\n# Initializing dataset and dataloader\ndataset = PCamDataset(labels_df, data_dir, transform=transform)\n# Adjust the batch size and num_workers to avoid the pin_memory issue\ndataloader = DataLoader(dataset, batch_size=128, shuffle=False, num_workers=5, pin_memory=False, persistent_workers=True)\n# List to store the IDs of non-noisy images\nvalid_image_ids = []\n\n# Using the Progress bar to monitor the progress of the loop\ntotal_images = len(labels_df)\nwith tqdm(total=total_images, desc=\"Processing Images\") as pbar:\n    for batch in dataloader:\n        img_ids, imgs = batch\n        for img_id, img in zip(img_ids, imgs):\n            img_tensor = img.to(device)\n            \n            if not is_noisy_image(img_tensor, threshold_percentage):\n                valid_image_ids.append(img_id)\n        \n        pbar.update(len(img_ids))\n\n# Filtering out the labels of the valid images\nclean_labels_df = labels_df[labels_df['id'].isin(valid_image_ids)]\n\n# Saving the cleaned labels file\nclean_labels_df.to_csv('clean_train_labels.csv', index=False)\n\nprint(f\"Total valid images: {len(valid_image_ids)}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-09-24T23:53:41.005826Z","iopub.execute_input":"2024-09-24T23:53:41.006221Z","iopub.status.idle":"2024-09-24T23:59:17.130817Z","shell.execute_reply.started":"2024-09-24T23:53:41.006185Z","shell.execute_reply":"2024-09-24T23:59:17.129596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nimport os\nimport torch\nfrom torch.utils.data import DataLoader, Dataset\nfrom torchvision import transforms\nfrom tqdm import tqdm\n\nclass CustomDataset(Dataset):\n    def __init__(self, labels_df, data_dir, transform):\n        self.labels_df = labels_df\n        self.data_dir = data_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.labels_df)\n\n    def __getitem__(self, idx):\n        img_id = self.labels_df.iloc[idx]['id']\n        label = self.labels_df.iloc[idx]['label']\n        \n        img_path = os.path.join(self.data_dir, img_id + '.tif')\n        image = Image.open(img_path).convert('RGB')\n        \n        if self.transform:\n            image = self.transform(image)\n        \n        return image, label\n\n# Define the transform\ntransform = transforms.Compose([\n    transforms.ToTensor()\n])\n\n# Create dataset\ndataset = CustomDataset(labels_df, data_dir, transform=transform)\n\n# Set a smaller batch size\nbatch_size = 32\n\n# Create DataLoader with smaller batch size\ndataloader = DataLoader(dataset, batch_size=batch_size, shuffle=True, num_workers=2)\n\n# Check the shape of a single cropped image and total batch with progress bar\nfor images, labels in tqdm(dataloader, desc=\"Loading images\", unit=\"batch\"):\n    print(\"Shape of single cropped image:\", images[0].shape) \n    print(\"Shape of full image batch:\", images.shape)        \n    break\n","metadata":{"execution":{"iopub.status.busy":"2024-09-24T18:44:44.401788Z","iopub.execute_input":"2024-09-24T18:44:44.402685Z","iopub.status.idle":"2024-09-24T18:44:44.737909Z","shell.execute_reply.started":"2024-09-24T18:44:44.402633Z","shell.execute_reply":"2024-09-24T18:44:44.736651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\n\n# Function to compute mean and std\ndef calculate_mean_std(dataloader):\n    # Initialize variables\n    mean = 0.0\n    std = 0.0\n    total_images = 0\n\n    for images, _ in dataloader:\n        # Calculate batch mean and std\n        batch_mean = images.mean(dim=[0, 2, 3])  # Mean for each channel\n        batch_std = images.std(dim=[0, 2, 3])    # Std for each channel\n\n        # Update the total images count\n        total_images += images.size(0)\n\n        # Update mean and std\n        mean += batch_mean * images.size(0)\n        std += batch_std * images.size(0)\n\n    # Finalize mean and std\n    mean /= total_images\n    std /= total_images\n\n    return mean, std\n\n# Calculate mean and std for the dataset\nmean, std = calculate_mean_std(dataloader)\n\nprint(f'Mean: {mean}')\nprint(f'Std: {std}')\n","metadata":{"execution":{"iopub.status.busy":"2024-09-24T18:45:00.219628Z","iopub.execute_input":"2024-09-24T18:45:00.22013Z","iopub.status.idle":"2024-09-24T18:49:48.472342Z","shell.execute_reply.started":"2024-09-24T18:45:00.220078Z","shell.execute_reply":"2024-09-24T18:49:48.470979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\n\n# Move the file to a writable directory\nshutil.move('clean_train_labels.csv', '/kaggle/working/clean_train_labels.csv')\n","metadata":{"execution":{"iopub.status.busy":"2024-09-24T18:51:54.31557Z","iopub.execute_input":"2024-09-24T18:51:54.316544Z","iopub.status.idle":"2024-09-24T18:51:54.324554Z","shell.execute_reply.started":"2024-09-24T18:51:54.316487Z","shell.execute_reply":"2024-09-24T18:51:54.323465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Function to display a noisy image inline using matplotlib\ndef display_noisy_image_inline(image_tensor, img_id):\n    # Convert the tensor back to PIL image\n    image_pil = transforms.ToPILImage()(image_tensor.cpu())\n    \n    # Display the image\n    plt.imshow(image_pil)\n    plt.title(f\"Image ID: {img_id}\")\n    plt.axis('off')\n    plt.show()\n\n# Display the first 5 noisy images inline\nfor i in range(min(5, len(noisy_images))):\n    img_id, img_tensor = noisy_images[i]\n    display_noisy_image_inline(img_tensor, img_id)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-24T18:52:01.022425Z","iopub.execute_input":"2024-09-24T18:52:01.022911Z","iopub.status.idle":"2024-09-24T18:52:01.505785Z","shell.execute_reply.started":"2024-09-24T18:52:01.022865Z","shell.execute_reply":"2024-09-24T18:52:01.504069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom PIL import Image\nfrom sklearn.model_selection import train_test_split\nimport torch\nfrom torch.utils.data import DataLoader, Dataset\nfrom torchvision import transforms\n\n# Custom Dataset class for handling PCam data\nclass PCamDataset(Dataset):\n    def __init__(self, labels_df, data_dir, transform=None):\n        self.labels_df = labels_df\n        self.data_dir = data_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.labels_df)\n\n    def __getitem__(self, idx):\n        img_id = self.labels_df.iloc[idx]['id']\n        img_path = os.path.join(self.data_dir, img_id + '.tif')\n        img = Image.open(img_path).convert('RGB')\n        label = self.labels_df.iloc[idx]['label']  # Assuming 'label' column exists in the DataFrame\n        if self.transform:\n            img = self.transform(img)\n        return img, label\n\n# Load clean labels DataFrame\nclean_labels_df = pd.read_csv('/kaggle/working/clean_train_labels.csv')\n\n# Define the path to the image data\ndata_dir = '/kaggle/input//kaggle/input/histopathologic-cancer-detection/train/'  # Update this path as needed\n\n# 70/10/20 Split: Split dataset into train, validation, and test sets\ntrain_df, temp_df = train_test_split(clean_labels_df, test_size=0.30, stratify=clean_labels_df['label'], random_state=42)\nval_df, test_df = train_test_split(temp_df, test_size=0.33, stratify=temp_df['label'], random_state=42)\n\n#print(f\"Train size: {len(train_df)}, Validation size: {len(val_df)}, Test size: {len(test_df)}\")\n\n# Define transforms (without data augmentation as requested)\nbasic_transform = transforms.Compose([\n    transforms.ToTensor()\n])\n\n# Create datasets\ntrain_dataset = PCamDataset(train_df, data_dir, transform=basic_transform)\nval_dataset = PCamDataset(val_df, data_dir, transform=basic_transform)\ntest_dataset = PCamDataset(test_df, data_dir, transform=basic_transform)\n\n# Define batch size\nbatch_size = 64\n\n# Create DataLoaders\ntrain_loader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True, num_workers=4)\nval_loader = DataLoader(val_dataset, batch_size=batch_size, shuffle=False, num_workers=4)\ntest_loader = DataLoader(test_dataset, batch_size=batch_size, shuffle=False, num_workers=4)\n\nprint(\"DataLoaders created for train, validation, and test sets.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-09-24T18:52:22.153397Z","iopub.execute_input":"2024-09-24T18:52:22.154676Z","iopub.status.idle":"2024-09-24T18:52:22.598257Z","shell.execute_reply.started":"2024-09-24T18:52:22.154627Z","shell.execute_reply":"2024-09-24T18:52:22.596969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom PIL import Image\nfrom sklearn.model_selection import train_test_split\nimport torch\nfrom torch.utils.data import DataLoader, Dataset\nfrom torchvision import transforms\n\n# Custom Dataset class for handling PCam data\nclass PCamDataset(Dataset):\n    def __init__(self, labels_df, data_dir, transform):\n        self.labels_df = labels_df\n        self.data_dir = data_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.labels_df)\n\n    def __getitem__(self, idx):\n        img_id = self.labels_df.iloc[idx]['id']\n        img_path = os.path.join(self.data_dir, img_id + '.tif')\n        img = Image.open(img_path).convert('RGB')\n        label = self.labels_df.iloc[idx]['label']  # Assuming 'label' column exists in the DataFrame\n        if self.transform:\n            img = self.transform(img)\n        return img, label\n\n# Load clean labels DataFrame\nclean_labels_df = pd.read_csv('/kaggle/working/clean_train_labels.csv')\n\n# Define the path to the image data\ndata_dir = '/kaggle/input/histopathologic-cancer-detection/train/'  # Update this path as needed\n\n# 70/10/20 Split: Split dataset into train, validation, and test sets\ntrain_df, temp_df = train_test_split(clean_labels_df, test_size=0.30, stratify=clean_labels_df['label'], random_state=42)\nval_df, test_df = train_test_split(temp_df, test_size=0.33, stratify=temp_df['label'], random_state=42)\n\n# Define transforms (with normalization)\nbasic_transform = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.7025, 0.5463, 0.6965], std=[0.2373, 0.2799, 0.2147])  # Normalization\n])\n\n# Create datasets\ntrain_dataset = PCamDataset(train_df, data_dir, transform=basic_transform)\nval_dataset = PCamDataset(val_df, data_dir, transform=basic_transform)\ntest_dataset = PCamDataset(test_df, data_dir, transform=basic_transform)\n\n# Define batch size\nbatch_size = 64\n\n# Create DataLoaders\ntrain_loader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True, num_workers=4)\nval_loader = DataLoader(val_dataset, batch_size=batch_size, shuffle=False, num_workers=4, drop_last=True)\ntest_loader = DataLoader(test_dataset, batch_size=batch_size, shuffle=False, num_workers=4, drop_last=True)\n\nprint(\"DataLoaders created for train, validation, and test sets.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-09-24T20:21:22.278206Z","iopub.execute_input":"2024-09-24T20:21:22.27913Z","iopub.status.idle":"2024-09-24T20:21:23.393751Z","shell.execute_reply.started":"2024-09-24T20:21:22.279074Z","shell.execute_reply":"2024-09-24T20:21:23.392691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''To provide a rough estimate, let’s assume that approximately 1 in 5 Americans develop skin cancer by age 70 (as stated by the Skin Cancer Foundation).This means that around 20% of the population will develop skin cancer at some point in their lifetime.'''","metadata":{"execution":{"iopub.status.busy":"2024-09-24T20:21:27.090443Z","iopub.execute_input":"2024-09-24T20:21:27.090833Z","iopub.status.idle":"2024-09-24T20:21:27.098354Z","shell.execute_reply.started":"2024-09-24T20:21:27.090797Z","shell.execute_reply":"2024-09-24T20:21:27.097207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nfrom tqdm import tqdm  # Import tqdm for progress bar\n\n# Define label values\ncancerous_label = 1  # Adjust this according to your dataset\nnon_cancerous_label = 0  # Adjust this according to your dataset\n\ndef check_distribution_and_shape(data_loader, cancerous_label):\n    cancer_count = 0\n    non_cancer_count = 0\n    total_count = 0\n\n    for _, labels in tqdm(data_loader, desc=\"Processing\", unit=\"batch\"):\n        cancer_count += (labels == cancerous_label).sum().item()\n        non_cancer_count += (labels == non_cancerous_label).sum().item()\n        total_count += labels.size(0)\n\n    return cancer_count, non_cancer_count, total_count\n\n# Set num_workers to 0 for debugging\nval_loader = DataLoader(val_dataset, batch_size=batch_size, shuffle=False, num_workers=0, drop_last=True)\ntest_loader = DataLoader(test_dataset, batch_size=batch_size, shuffle=False, num_workers=0, drop_last=True)\n\n# Check distribution for validation loader\nval_cancer_count, val_non_cancer_count, val_total_count = check_distribution_and_shape(val_loader, cancerous_label)\nprint(f\"Validation Loader Distribution - Cancerous: {val_cancer_count}, Non-Cancerous: {val_non_cancer_count}, Total: {val_total_count}\")\n\n# Check distribution for test loader\ntest_cancer_count, test_non_cancer_count, test_total_count = check_distribution_and_shape(test_loader, cancerous_label)\nprint(f\"Test Loader Distribution - Cancerous: {test_cancer_count}, Non-Cancerous: {test_non_cancer_count}, Total: {test_total_count}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-09-24T20:22:20.880554Z","iopub.execute_input":"2024-09-24T20:22:20.881032Z","iopub.status.idle":"2024-09-24T20:25:09.056252Z","shell.execute_reply.started":"2024-09-24T20:22:20.880968Z","shell.execute_reply":"2024-09-24T20:25:09.055098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nfrom torch.utils.data import DataLoader, Subset\nimport random\nfrom tqdm import tqdm  # Import tqdm for progress bar\n\n# Function to create a balanced data loader with 20% cancerous and 80% non-cancerous samples\ndef create_balanced_loader_with_progress(data_loader, cancerous_label, non_cancerous_label, cancerous_ratio=0.2):\n    cancerous_indices = []\n    non_cancerous_indices = []\n    \n    # Use tqdm for the progress bar while iterating over the dataset\n    print(\"Separating cancerous and non-cancerous samples...\")\n    for idx, (images, labels) in tqdm(enumerate(data_loader.dataset), total=len(data_loader.dataset), desc=\"Processing\"):\n        if labels == cancerous_label:\n            cancerous_indices.append(idx)\n        else:\n            non_cancerous_indices.append(idx)\n    \n    # Shuffle the indices\n    random.shuffle(cancerous_indices)\n    random.shuffle(non_cancerous_indices)\n    \n    # Determine number of samples for the new loader\n    num_cancerous = int(len(cancerous_indices) * cancerous_ratio)\n    num_non_cancerous = int(len(non_cancerous_indices) * (1 - cancerous_ratio))\n    \n    # Select the required amount of cancerous and non-cancerous samples\n    selected_cancerous_indices = cancerous_indices[:num_cancerous]\n    selected_non_cancerous_indices = non_cancerous_indices[:num_non_cancerous]\n    \n    # Combine the selected indices\n    new_indices = selected_cancerous_indices + selected_non_cancerous_indices\n    random.shuffle(new_indices)\n    \n    # Create the new dataset and data loader\n    new_subset = Subset(data_loader.dataset, new_indices)\n    new_loader = DataLoader(new_subset, batch_size=data_loader.batch_size, shuffle=True)\n    \n    return new_loader\n\n# Function to check the distribution of cancerous and non-cancerous samples in a data loader\ndef check_distribution_with_progress(data_loader, cancerous_label):\n    cancer_count = 0\n    non_cancer_count = 0\n    total_count = 0\n    \n    print(\"Checking distribution in the new data loader...\")\n    \n    # Use tqdm for the progress bar\n    for images, labels in tqdm(data_loader, desc=\"Checking\"):\n        cancer_count += (labels == cancerous_label).sum().item()\n        non_cancer_count += (labels != cancerous_label).sum().item()\n        total_count += len(labels)\n    \n    return cancer_count, non_cancer_count, total_count\n\n# Define your cancerous and non-cancerous labels\ncancerous_label = 1\nnon_cancerous_label = 0\n\n# Create the new test data loader with 20% cancerous and 80% non-cancerous images\nnew_test_loader = create_balanced_loader_with_progress(test_loader, cancerous_label, non_cancerous_label, cancerous_ratio=0.2)\n\n# Check the distribution of the new test loader\ntest_cancer_count, test_non_cancer_count, test_total_count = check_distribution_with_progress(new_test_loader, cancerous_label)\n\n# Print the distribution\nprint(f\"New Test Loader Distribution:\")\nprint(f\"Cancerous: {test_cancer_count}, Non-Cancerous: {test_non_cancer_count}, Total: {test_total_count}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-09-24T20:25:21.908363Z","iopub.execute_input":"2024-09-24T20:25:21.909289Z","iopub.status.idle":"2024-09-24T20:26:32.095804Z","shell.execute_reply.started":"2024-09-24T20:25:21.90924Z","shell.execute_reply":"2024-09-24T20:26:32.094691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nfrom torch.utils.data import DataLoader, Subset\nimport random\nfrom tqdm import tqdm  # Import tqdm for progress bar\n\n# Function to create a balanced data loader with 20% cancerous and 80% non-cancerous samples\ndef create_balanced_loader_with_progress(data_loader, cancerous_label, non_cancerous_label, cancerous_ratio=0.2):\n    cancerous_indices = []\n    non_cancerous_indices = []\n    \n    # Use tqdm for the progress bar while iterating over the dataset\n    print(\"Separating cancerous and non-cancerous samples...\")\n    for idx, (images, labels) in tqdm(enumerate(data_loader.dataset), total=len(data_loader.dataset), desc=\"Processing\"):\n        if labels == cancerous_label:\n            cancerous_indices.append(idx)\n        else:\n            non_cancerous_indices.append(idx)\n    \n    # Shuffle the indices\n    random.shuffle(cancerous_indices)\n    random.shuffle(non_cancerous_indices)\n    \n    # Determine number of samples for the new loader\n    num_cancerous = int(len(cancerous_indices) * cancerous_ratio)\n    num_non_cancerous = int(len(non_cancerous_indices) * (1 - cancerous_ratio))\n    \n    # Select the required amount of cancerous and non-cancerous samples\n    selected_cancerous_indices = cancerous_indices[:num_cancerous]\n    selected_non_cancerous_indices = non_cancerous_indices[:num_non_cancerous]\n    \n    # Combine the selected indices\n    new_indices = selected_cancerous_indices + selected_non_cancerous_indices\n    random.shuffle(new_indices)\n    \n    # Create the new dataset and data loader\n    new_subset = Subset(data_loader.dataset, new_indices)\n    new_loader = DataLoader(new_subset, batch_size=data_loader.batch_size, shuffle=True, drop_last=True)\n    \n    return new_loader\n\n# Function to check the distribution of cancerous and non-cancerous samples in a data loader\ndef check_distribution_with_progress(data_loader, cancerous_label):\n    cancer_count = 0\n    non_cancer_count = 0\n    total_count = 0\n    \n    print(\"Checking distribution in the new data loader...\")\n    \n    # Use tqdm for the progress bar\n    for images, labels in tqdm(data_loader, desc=\"Checking\"):\n        cancer_count += (labels == cancerous_label).sum().item()\n        non_cancer_count += (labels != cancerous_label).sum().item()\n        total_count += len(labels)\n    \n    return cancer_count, non_cancer_count, total_count\n\n# Define your cancerous and non-cancerous labels\ncancerous_label = 1\nnon_cancerous_label = 0\n\n# --- For Validation Loader ---\n# Create the new validation data loader with 20% cancerous and 80% non-cancerous images\nnew_val_loader = create_balanced_loader_with_progress(val_loader, cancerous_label, non_cancerous_label, cancerous_ratio=0.2)\n\n# Check the distribution of the new validation loader\nval_cancer_count, val_non_cancer_count, val_total_count = check_distribution_with_progress(new_val_loader, cancerous_label)\n\n# Print the distribution for validation loader\nprint(f\"New Validation Loader Distribution:\")\nprint(f\"Cancerous: {val_cancer_count}, Non-Cancerous: {val_non_cancer_count}, Total: {val_total_count}\")\n\n\n# --- For Test Loader ---\n# Create the new test data loader with 20% cancerous and 80% non-cancerous images\nnew_test_loader = create_balanced_loader_with_progress(test_loader, cancerous_label, non_cancerous_label, cancerous_ratio=0.2)\n\n# Check the distribution of the new test loader\ntest_cancer_count, test_non_cancer_count, test_total_count = check_distribution_with_progress(new_test_loader, cancerous_label)\n\n# Print the distribution for test loader\nprint(f\"New Test Loader Distribution:\")\nprint(f\"Cancerous: {test_cancer_count}, Non-Cancerous: {test_non_cancer_count}, Total: {test_total_count}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-09-24T20:27:53.472141Z","iopub.execute_input":"2024-09-24T20:27:53.473147Z","iopub.status.idle":"2024-09-24T20:31:22.450197Z","shell.execute_reply.started":"2024-09-24T20:27:53.473096Z","shell.execute_reply":"2024-09-24T20:31:22.449067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch.set_printoptions(edgeitems=2, threshold=50)\n!pip install timm\nimport timm","metadata":{"execution":{"iopub.status.busy":"2024-09-24T20:31:27.186471Z","iopub.execute_input":"2024-09-24T20:31:27.186938Z","iopub.status.idle":"2024-09-24T20:31:41.811126Z","shell.execute_reply.started":"2024-09-24T20:31:27.186893Z","shell.execute_reply":"2024-09-24T20:31:41.809586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"timm.list_models(pretrained=True)","metadata":{"execution":{"iopub.status.busy":"2024-09-24T20:31:45.757308Z","iopub.execute_input":"2024-09-24T20:31:45.758202Z","iopub.status.idle":"2024-09-24T20:31:45.801326Z","shell.execute_reply.started":"2024-09-24T20:31:45.758142Z","shell.execute_reply":"2024-09-24T20:31:45.800185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = timm.create_model('swin_tiny_patch4_window7_224', pretrained = True)","metadata":{"execution":{"iopub.status.busy":"2024-09-24T20:31:47.593831Z","iopub.execute_input":"2024-09-24T20:31:47.594852Z","iopub.status.idle":"2024-09-24T20:31:48.30268Z","shell.execute_reply.started":"2024-09-24T20:31:47.594803Z","shell.execute_reply":"2024-09-24T20:31:48.301767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport timm\n\nclass CustomSwinTransformer(nn.Module):\n    def __init__(self, num_classes=2, img_size=96, patch_size=32):\n        super(CustomSwinTransformer, self).__init__()\n        self.model = timm.create_model('swin_tiny_patch4_window7_224', pretrained=True, num_classes=num_classes)\n        self.img_size = img_size\n        self.patch_size = patch_size\n\n        # Adjusting the input size of the model\n        self.model.patch_embed.img_size = (img_size, img_size)  # Set image size for model\n        self.model.patch_embed.patches = (img_size // patch_size, img_size // patch_size)\n\n    def forward(self, x):\n        return self.model(x)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-24T20:31:50.160323Z","iopub.execute_input":"2024-09-24T20:31:50.160759Z","iopub.status.idle":"2024-09-24T20:31:50.169353Z","shell.execute_reply.started":"2024-09-24T20:31:50.160718Z","shell.execute_reply":"2024-09-24T20:31:50.168065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\n\ndef train_model(model, train_loader, criterion, optimizer, num_epochs, device):\n    model.train()  # Set the model to training mode\n    for epoch in range(num_epochs):\n        running_loss = 0.0\n        correct = 0\n        total = 0\n\n        # Iterate through the training data\n        for batch_idx, (images, labels) in enumerate(tqdm(train_loader, desc=f'Training Epoch {epoch+1}/{num_epochs}', unit='batch')):\n            # Debugging: Print batch info to detect hanging points\n            print(f\"Processing batch {batch_idx + 1}/{len(train_loader)}\")\n\n            images, labels = images.to(device), labels.to(device)  # Move data to device\n            \n            optimizer.zero_grad()  # Zero the gradients\n            outputs = model(images)  # Forward pass\n            \n            loss = criterion(outputs, labels)  # Compute the loss\n            if torch.isnan(loss).any() or torch.isinf(loss).any():\n                print(f\"Warning: Detected NaN or Inf loss at batch {batch_idx}\")\n                continue  # Skip the batch if there's an invalid loss value\n\n            loss.backward()  # Backward pass\n            optimizer.step()  # Update the weights\n\n            running_loss += loss.item()  # Accumulate loss\n            _, predicted = torch.max(outputs.data, 1)\n            total += labels.size(0)\n            correct += (predicted == labels).sum().item()\n\n        epoch_loss = running_loss / len(train_loader)  # Average loss\n        epoch_accuracy = correct / total  # Accuracy for the epoch\n\n        print(f'Epoch [{epoch+1}/{num_epochs}], Loss: {epoch_loss:.4f}, Accuracy: {epoch_accuracy:.4f}')\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import roc_curve, auc\nfrom tqdm import tqdm\n\ndef validate_model_with_roc(model, val_loader, criterion, device):\n    model.eval()  # Set the model to evaluation mode\n    running_loss = 0.0\n    correct = 0\n    total = 0\n\n    all_labels = []\n    all_preds = []\n\n    with torch.no_grad():  # Disable gradient computation\n        for batch_idx, (images, labels) in enumerate(tqdm(val_loader, desc='Validation', unit='batch')):\n            print(f\"Processing batch {batch_idx + 1}/{len(val_loader)}\")  # Debugging print to track progress\n\n            images, labels = images.to(device), labels.to(device)  # Move data to device\n\n            outputs = model(images)  # Forward pass\n            loss = criterion(outputs, labels)  # Compute the loss\n\n            if torch.isnan(loss).any() or torch.isinf(loss).any():\n                print(f\"Warning: Detected NaN or Inf loss at batch {batch_idx}, skipping this batch.\")\n                continue  # Skip the batch if loss contains invalid values\n\n            running_loss += loss.item()  # Accumulate loss\n\n            # Get predictions and compute accuracy\n            _, predicted = torch.max(outputs, 1)\n            total += labels.size(0)  # Increment total with batch size\n            correct += (predicted == labels).sum().item()  # Increment correct predictions\n\n            all_labels.extend(labels.cpu().numpy())\n            all_preds.extend(torch.softmax(outputs, dim=1)[:, 1].cpu().numpy())  # Get probabilities for class 1\n\n    # Ensure there's no division by zero\n    val_loss = running_loss / len(val_loader) if len(val_loader) > 0 else float('inf')  # Average loss\n    val_accuracy = correct / total if total > 0 else 0  # Accuracy for the validation set\n\n    print(f'Validation Loss: {val_loss:.4f}, Accuracy: {val_accuracy:.4f}')\n\n    # Check if there are predictions and labels\n    if len(all_labels) == 0 or len(all_preds) == 0:\n        print(\"Warning: No data for ROC computation. Skipping ROC curve plot.\")\n        return\n\n    # Convert lists to numpy arrays\n    all_labels = np.array(all_labels)\n    all_preds = np.array(all_preds)\n\n    # Calculate ROC curve\n    fpr, tpr, thresholds = roc_curve(all_labels, all_preds)\n    roc_auc = auc(fpr, tpr)\n\n    # Plot ROC curve\n    plt.figure()\n    plt.plot(fpr, tpr, color='blue', label='ROC curve (area = %0.2f)' % roc_auc)\n    plt.plot([0, 1], [0, 1], color='red', linestyle='--')  # Diagonal line\n    plt.xlim([0.0, 1.0])\n    plt.ylim([0.0, 1.05])\n    plt.xlabel('False Positive Rate')\n    plt.ylabel('True Positive Rate')\n    plt.title('Receiver Operating Characteristic (ROC) Curve')\n    plt.legend(loc='lower right')\n    plt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize model, criterion, and optimizer\nmodel = CustomSwinTransformer(num_classes=2).to(device)\ncriterion = nn.CrossEntropyLoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=1e-4)\n\n# Define number of epochs\nnum_epochs = 10\n\n# Training and Validation loop\nfor epoch in range(num_epochs):\n    train_model(model, train_loader, criterion, optimizer, num_epochs=1, device=device)\n    validate_model_with_roc(model, new_val_loader, criterion, device=device)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}