{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":117682,"databundleVersionId":15062069},{"sourceType":"datasetVersion","sourceId":14832123,"datasetId":1608934,"databundleVersionId":15690141}],"dockerImageVersionId":31287,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Basic DataLoader Demo using Brain Tumor MRI Dataset\n\nDataset URL: https://www.kaggle.com/datasets/masoudnickparvar/brain-tumor-mri-dataset\n\n","metadata":{}},{"cell_type":"markdown","source":"Note: Don't go for the \"run all option\" since this notebook is intented to demonstrate the working of standard dataloaders on large datasets (which means there will be a crash in the end). Take this one slow, run it cell by sell to see how it goes!","metadata":{}},{"cell_type":"code","source":"import os\nimport torch\nimport torch.nn as nn\nfrom torchvision import transforms\nfrom torchvision import datasets\nfrom torch.utils.data import Dataset, DataLoader\nfrom PIL import Image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-03T17:33:12.480335Z","iopub.execute_input":"2026-03-03T17:33:12.481192Z","iopub.status.idle":"2026-03-03T17:33:12.484712Z","shell.execute_reply.started":"2026-03-03T17:33:12.481163Z","shell.execute_reply":"2026-03-03T17:33:12.484123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(f\"Using Device: {DEVICE}\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-03T17:33:12.493920Z","iopub.execute_input":"2026-03-03T17:33:12.494517Z","iopub.status.idle":"2026-03-03T17:33:12.580003Z","shell.execute_reply.started":"2026-03-03T17:33:12.494495Z","shell.execute_reply":"2026-03-03T17:33:12.579192Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Technically we don't need this step for this demo since we are not running the model (or loading the data to the GPU). But, for the sake of consistency and avoiding future confusion, I will keep it here anyway","metadata":{}},{"cell_type":"code","source":"class BrainTumorDataset(Dataset):\n    def __init__(self, root_dir, transform=None):\n        \"\"\"\n        Initializes the dataset object. Takes path and transform as inputs\n        \"\"\"\n\n        \n        super().__init__() \n        \n        self.root_dir = root_dir\n        self.transform = transform\n        \n        # Manually map folder names to integers\n        self.classes = sorted(os.listdir(root_dir))\n        self.class_to_idx = {cls_name: i for i, cls_name in enumerate(self.classes)}\n        \n        # 3. Build the file list\n        self.images = []\n        for cls_name in self.classes:\n            cls_path = os.path.join(root_dir, cls_name)\n            for img_name in os.listdir(cls_path):\n                self.images.append((os.path.join(cls_path, img_name), self.class_to_idx[cls_name]))\n\n    def __len__(self):\n        # Returns the number of images in the dataset\n        return len(self.images)\n\n    def __getitem__(self, idx):\n        \"\"\"\n        Grabs a sepcific sample, loads it and applies transforms on it\n        \"\"\"\n\n        img_path, label = self.images[idx]\n        image = Image.open(img_path).convert(\"RGB\")\n        \n        if self.transform:\n            image = self.transform(image)\n            \n        return image, label\n\n# Add your path here. Loading in kaggle directly\ndata_path = '/kaggle/input/datasets/masoudnickparvar/brain-tumor-mri-dataset'\n\n# Adding some basic resizing transform before converting it to tensor\nmri_transforms = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n])\n\n# Instantiate Datasets using the path of the data\ntrain_dataset = BrainTumorDataset(os.path.join(data_path, 'Training'), transform=mri_transforms)\ntest_dataset = BrainTumorDataset(os.path.join(data_path, 'Testing'), transform=mri_transforms)\n\n# Finally the DataLoader to create and load batches for model training. \n# Note how with a smaller dataset, loading a batch size of 32 is also feasible!\ntrain_loader = DataLoader(\n    train_dataset, \n    batch_size=32, \n    shuffle=True, \n    num_workers=os.cpu_count(), \n    pin_memory=True\n)\n\ntest_loader = DataLoader(\n    test_dataset, \n    batch_size=32, \n    shuffle=False, \n    num_workers=os.cpu_count(), \n    pin_memory=True\n)\n\nprint(f\"Loaded {len(train_dataset)} training images across classes: {train_dataset.classes}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-03T17:33:12.581508Z","iopub.execute_input":"2026-03-03T17:33:12.581756Z","iopub.status.idle":"2026-03-03T17:33:12.725247Z","shell.execute_reply.started":"2026-03-03T17:33:12.581735Z","shell.execute_reply":"2026-03-03T17:33:12.724505Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"In PyTorch, the Dataset and DataLoader work together as a two-stage relay system to move data from your \"pocket\" (storage) to the \"unfiltered\" engine (GPU).\n\n1. The Dataset Class (The \"What\")\nThe Dataset is a container that holds your data and defines how to access a single sample.\n\nResponsibility: It stores the samples and their corresponding labels.\n\nKey Logic: It implements __getitem__, which tells PyTorch exactly how to grab one image and its label from a specific index.\n\nAnalogy: Think of this as the Library—it holds all the individual books (data) and knows exactly where each one is on the shelf.\n\n2. The DataLoader Class (The \"How\")\nThe DataLoader is a wrapper that manages the flow of data.\n\nResponsibility: It groups individual samples into batches, shuffles them to prevent the model from memorizing patterns, and manages parallel processing.\n\nHardware Efficiency: It uses num_workers to load data in the background (using your CPU/RAM) so the GPU is never \"starved\" of data.\n\nAnalogy: Think of this as the Delivery Truck—it takes books from the library, packs them into boxes (batches), and drives them to the reader (model) at high speed.\n\nThe Dataset class typically stores the transform logic and applies it to each individual sample within its __getitem__ method when that sample is requested.\n\nGemini said\nTo make this \"unfiltered\" and \"pocket-friendly,\" we will manually implement the Dataset class. This gives you total control over how images are read and transformed, which is essential if you ever move beyond simple folder structures.\n\nWhy we used these specific methods:\n__init__: We scan the folders once at the start. This saves time during training because we aren't constantly checking the disk for file lists.\n\n__getitem__: This is where the RAM workload happens. We only open the image file at the exact moment the DataLoader asks for it, keeping your memory usage low.\n\nImage.open(...).convert(\"RGB\"): Ensures that even if an MRI is grayscale, it matches the 3-channel input expected by most pre-trained models (like ResNet).","metadata":{}},{"cell_type":"code","source":"import os\nimport torch\nfrom torchvision import datasets, transforms\nfrom torch.utils.data import DataLoader\n\n#Define the path\ndata_path = '/kaggle/input/datasets/masoudnickparvar/brain-tumor-mri-dataset'\ntrain_path = os.path.join(data_path, 'Training')\ntest_path = os.path.join(data_path, 'Testing')\n\n# Add some transforms\ntransform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n])\n\n# Using Imagefolder which is it is a pre-built subclass of the Dataset class\ntrain_dataset = datasets.ImageFolder(root=train_path, transform=transform)\ntest_dataset = datasets.ImageFolder(root=test_path, transform=transform)\n\n# Dataloaders\ntrain_loader = DataLoader(\n    train_dataset, \n    batch_size=32, \n    shuffle=True, \n    num_workers=os.cpu_count(),\n    pin_memory=True\n)\n\ntest_loader = DataLoader(\n    test_dataset, \n    batch_size=32, \n    shuffle=False, \n    num_workers=os.cpu_count(),\n    pin_memory=True\n)\n\nprint(f\"Classes found: {train_dataset.classes}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-03T17:33:12.726268Z","iopub.execute_input":"2026-03-03T17:33:12.726572Z","iopub.status.idle":"2026-03-03T17:33:14.968918Z","shell.execute_reply.started":"2026-03-03T17:33:12.726540Z","shell.execute_reply":"2026-03-03T17:33:14.967839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install imagecodecs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-03T18:08:35.210477Z","iopub.execute_input":"2026-03-03T18:08:35.210674Z","iopub.status.idle":"2026-03-03T18:08:40.464329Z","shell.execute_reply.started":"2026-03-03T18:08:35.210653Z","shell.execute_reply":"2026-03-03T18:08:40.463488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport torch\nimport torch.nn.functional as F\nimport tifffile as tiff\nimport numpy as np\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\nimport time \n\n\nclass Resize3D:\n    \"\"\"\n    PyTorch doesn't handle 3D volumes :(\n    So this resize option will take care of the resizing\n    Note most volumes in the dataset are (320,320,320) and we are resizing it to (256, 256, 256) here\n    \"\"\"\n    def __init__(self, size=(256, 256, 256)):\n        self.size = size\n    def __call__(self, volume_tensor):\n        # As per pytorch, the dataset has to be 5D arranged as [Batch, Channel, Depth, Height, Width]\n        volume_tensor = volume_tensor.unsqueeze(0).unsqueeze(0)  ## Double unsqueeze to add the extra two dimensions\n        resized_volume = F.interpolate(\n            volume_tensor, size=self.size, mode='trilinear', align_corners=False\n        )\n        return resized_volume.squeeze(0).squeeze(0)  ### Squeezing it batch, dataloader will add the batch dimension later \n\nclass VesuviusDataset3D(Dataset):\n    \"\"\"\n    Our main dataset class. Set the paths, return length (fixed to 1 here). Methods for loading volume and \n    creating tensors and finally __getitem__ to return a single sample with custom transforms\n    \"\"\"\n    def __init__(self, fragment_path, volume_transform=None, mask_transform=None):\n        super().__init__()\n        self.volume_dir = os.path.join(fragment_path, 'train_images')\n        self.labels_dir = os.path.join(fragment_path, 'train_labels')\n        \n        # Filter for .tif files and ensure we only take the first 320 (or however many you need)\n        self.filenames = sorted([f for f in os.listdir(self.volume_dir) if f.endswith('.tif')])\n\n    def __len__(self):\n        ## Trust me, you won't be able to load than 1 in this approach!\n        return 1 \n\n    def _load_volume(self, directory, filenames):\n        #### No transforms, so we do it here\n        slices = []\n        for f in filenames:\n            img_path = os.path.join(directory, f)\n            img = tiff.imread(img_path)\n            slices.append(torch.from_numpy(img).float())  ### Converts to tensors        \n        # Stack along the depth dimension\n        return torch.stack(slices, dim=0) / 255.0\n\n    def __getitem__(self, idx):\n        # Load Volume and Mask manually to avoid size errors\n        volume_tensor = self._load_volume(self.volume_dir, self.filenames)\n        mask_tensor = self._load_volume(self.labels_dir, self.filenames)\n        \n        # Apply the resize\n        if self.volume_transform:\n            volume_tensor = self.volume_transform(volume_tensor)\n        if self.mask_transform:\n            mask_tensor = self.mask_transform(mask_tensor)            \n        return volume_tensor, mask_tensor\n\n\ndata_path = '/kaggle/input/competitions/vesuvius-challenge-surface-detection'\n\n## custom transforms\ntransform_3d = transforms.Compose([\n    Resize3D(size=(256, 256, 256))\n])\n\n# Instantiate Datasets using the path of the data\ntrain_dataset = VesuviusDataset3D(\n    data_path, \n    volume_transform=transform_3d, \n    mask_transform=transform_3d\n)\n\nimage_stack, mask_stack = train_dataset[0]\nprint(f\"Volume Shape: {image_stack.shape}\") \nprint(f\"Mask Shape: {mask_stack.shape}\")\n\n### And here's the dataloader\ntrain_loader = DataLoader(\n    train_dataset, \n    batch_size=1, \n    shuffle=True, \n    num_workers=2,    # Parallel threads to load slices\n    pin_memory=True   \n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-03T18:08:40.466294Z","iopub.execute_input":"2026-03-03T18:08:40.467037Z","execution_failed":"2026-03-03T18:11:10.008Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"To make this \"hardware and pocket-friendly,\" we will swap the EMNIST-specific loader for PyTorch's ImageFolder. This is the most efficient way to load data structured exactly like your MRI dataset, as it maps the folder names (glioma, meningioma, etc.) to class labels automatically. \nWhy this is the \"Unfiltered\" way to do it:\nZero Redundancy: ImageFolder eliminates the need for manual CSVs or label mapping files. If you add a new tumor type folder later, it just works.\n\nMemory Efficiency: By using DataLoader with pin_memory=True, you speed up the transfer from RAM to GPU VRAM, which is the primary bottleneck in MRI training.\n\nCPU Parallelism: num_workers=os.cpu_count() ensures your CPU is busy preparing the next batch while your GPU is crunching the current one—maximizing your hardware utilization.","metadata":{}},{"cell_type":"code","source":"# import tensorflow as tf\n# import os\n\n# # Define paths based on your directory structure\n# base_path = 'Brain Tumor MRI Dataset'\n# train_dir = os.path.join(base_path, 'Training')\n# test_dir = os.path.join(base_path, 'Testing')\n\n# # Hardware-friendly parameters\n# batch_size = 32  # Reduce to 16 or 8 if you run out of VRAM\n# img_size = (224, 224) \n\n# # Load Training Data\n# train_ds = tf.keras.utils.image_dataset_from_directory(\n#     train_dir,\n#     labels='inferred',\n#     label_mode='categorical', # or 'sparse' for integer labels\n#     batch_size=batch_size,\n#     image_size=img_size,\n#     shuffle=True\n# )\n\n# # Load Testing Data\n# test_ds = tf.keras.utils.image_dataset_from_directory(\n#     test_dir,\n#     labels='inferred',\n#     label_mode='categorical',\n#     batch_size=batch_size,\n#     image_size=img_size,\n#     shuffle=False\n# )\n\n# # The \"Pocket-Friendly\" Optimizer\n# # This ensures the CPU prepares Batch B while the GPU is processing Batch A\n# train_ds = train_ds.cache().prefetch(buffer_size=tf.data.AUTOTUNE)\n# test_ds = test_ds.cache().prefetch(buffer_size=tf.data.AUTOTUNE)  Tf version","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-03T17:33:14.972662Z","iopub.status.idle":"2026-03-03T17:33:14.972936Z","shell.execute_reply.started":"2026-03-03T17:33:14.972777Z","shell.execute_reply":"2026-03-03T17:33:14.972790Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-03T17:33:14.974077Z","iopub.status.idle":"2026-03-03T17:33:14.974389Z","shell.execute_reply.started":"2026-03-03T17:33:14.974262Z","shell.execute_reply":"2026-03-03T17:33:14.974281Z"}},"outputs":[],"execution_count":null}]}