{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":117682,"databundleVersionId":15062069},{"sourceType":"datasetVersion","sourceId":14832123,"datasetId":1608934,"databundleVersionId":15690141}],"dockerImageVersionId":31287,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Basic DataLoader Demo using Brain Tumor MRI Dataset\n\nDataset URL: https://www.kaggle.com/datasets/masoudnickparvar/brain-tumor-mri-dataset","metadata":{}},{"cell_type":"markdown","source":"Note: Don't go for the \"run all option\" since this notebook is intented to demonstrate the working of standard dataloaders on large datasets (which means there will be a crash in the end). Take this one slow, run it cell by sell to see how it goes!","metadata":{}},{"cell_type":"code","source":"import os\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3' \nimport tensorflow as tf\n\n\n# Check if GPUs are available\ngpus = tf.config.list_physical_devices('GPU')\n\nif gpus:\n    print(f\"Using Device: GPU ({len(gpus)} available)\")\n    # Optional: Display the specific name of the GPU\n    for gpu in gpus:\n        print(f\"  - {gpu.name}\")\nelse:\n    print(\"Using Device: CPU\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-15T07:15:12.573516Z","iopub.execute_input":"2026-04-15T07:15:12.573736Z","iopub.status.idle":"2026-04-15T07:15:51.843043Z","shell.execute_reply.started":"2026-04-15T07:15:12.573714Z","shell.execute_reply":"2026-04-15T07:15:51.842349Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Device management in Tensorflow is more automated than in PyTorch. While PyTorch requires you to explicitly send both the model and the data to the device using .to(DEVICE), TensorFlow typically detects and uses the GPU by default if it is available.","metadata":{}},{"cell_type":"code","source":"# Setup paths and classes\ndata_path = '/kaggle/input/datasets/masoudnickparvar/brain-tumor-mri-dataset/Training'\nclasses = sorted(os.listdir(data_path))\nclass_to_idx = {cls_name: i for i, cls_name in enumerate(classes)}\n# Build the list of file paths and labels\nfile_paths = []\nlabels = []\n\nfor cls_name in classes:\n    cls_path = os.path.join(data_path, cls_name)\n    for img_name in os.listdir(cls_path):\n        file_paths.append(os.path.join(cls_path, img_name))\n        labels.append(class_to_idx[cls_name])\n\n# Define the processing function (The \"Transform\")\ndef process_path(file_path, label):\n    # Load the raw data from the file as a string\n    img = tf.io.read_file(file_path)\n    # Decode jpeg/png to a uint8 tensor\n    img = tf.image.decode_jpeg(img, channels=3)\n    # Resize (equivalent to transforms.Resize)\n    img = tf.image.resize(img, [224, 224])\n    # Normalize to [0, 1] (equivalent to transforms.ToTensor)\n    img = tf.cast(img, tf.float32) / 255.0\n    return img, label\n\n# Create a dataset of file paths and labels\ndataset = tf.data.Dataset.from_tensor_slices((file_paths, labels))\n\n# Chain the operations\ntrain_ds = (\n    dataset\n    .shuffle(len(file_paths))                   # Shuffle the data\n    .map(process_path, num_parallel_calls=tf.data.AUTOTUNE) # Apply transforms\n    .batch(32)                                  # Create batches\n    .prefetch(tf.data.AUTOTUNE)                 # Prefetch for speed\n)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-04-15T07:15:51.844530Z","iopub.execute_input":"2026-04-15T07:15:51.845026Z","iopub.status.idle":"2026-04-15T07:15:52.225829Z","shell.execute_reply.started":"2026-04-15T07:15:51.844999Z","shell.execute_reply":"2026-04-15T07:15:52.225171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## just printing train_ds to see what it contains\n\ntrain_ds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-15T07:15:52.226694Z","iopub.execute_input":"2026-04-15T07:15:52.226955Z","iopub.status.idle":"2026-04-15T07:15:52.232174Z","shell.execute_reply.started":"2026-04-15T07:15:52.226933Z","shell.execute_reply":"2026-04-15T07:15:52.231497Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"That's it! Notice how similar it looks to PyTorch's Imagefolder? Since the images are already organized in folders, there's an even cleaner way to do this!","metadata":{}},{"cell_type":"code","source":"data_path = '/kaggle/input/datasets/masoudnickparvar/brain-tumor-mri-dataset/'\n\n# Load Training Dataset\ntrain_ds = tf.keras.utils.image_dataset_from_directory(\n    os.path.join(data_path, 'Training'),\n    labels='inferred',            # Automatically uses folder names as labels\n    label_mode='int',             # Maps labels to integers (0, 1, 2, 3)\n    class_names=None,             # Can specify list of strings to set order\n    color_mode='rgb',\n    batch_size=32,\n    image_size=(224, 224),        # Resizes images automatically\n    shuffle=True,\n    seed=42\n)\n\n# Load Testing Dataset\ntest_ds = tf.keras.utils.image_dataset_from_directory(\n    os.path.join(data_path, 'Testing'),\n    labels='inferred',\n    label_mode='int',\n    color_mode='rgb',\n    batch_size=32,\n    image_size=(224, 224),\n    shuffle=False                 # No need to shuffle test data\n)\n\n# This keeps images in memory after they're loaded off disk during the first epoch and prefetches batches \n# so the GPU doesn't wait for the CPU similar to pin_memory and num_workers\nAUTOTUNE = tf.data.AUTOTUNE\ntrain_ds = train_ds.cache().prefetch(buffer_size=AUTOTUNE)\ntest_ds = test_ds.cache().prefetch(buffer_size=AUTOTUNE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-15T07:15:52.232919Z","iopub.execute_input":"2026-04-15T07:15:52.233465Z","iopub.status.idle":"2026-04-15T07:16:00.330707Z","shell.execute_reply.started":"2026-04-15T07:15:52.233435Z","shell.execute_reply":"2026-04-15T07:16:00.330117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # How to add augmentation to the above\n# from tensorflow.keras import layers\n\n# # Define the augmentation block\n# data_augmentation = tf.keras.Sequential([\n#   layers.RandomFlip(\"horizontal\"),\n#   layers.RandomRotation(0.1),\n#   layers.RandomZoom(0.1),\n# ])\n\n# # Use it in your model\n# model = tf.keras.Sequential([\n#   layers.Input(shape=(224, 224, 3)),\n#   data_augmentation,    # Augmentation happens here (on the GPU)\n#   layers.Rescaling(1./255),\n#   # ... your conv layers ...\n# ])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-15T07:16:00.332583Z","iopub.execute_input":"2026-04-15T07:16:00.332810Z","iopub.status.idle":"2026-04-15T07:16:00.336738Z","shell.execute_reply.started":"2026-04-15T07:16:00.332789Z","shell.execute_reply":"2026-04-15T07:16:00.336119Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-15T07:16:00.337527Z","iopub.execute_input":"2026-04-15T07:16:00.337737Z","iopub.status.idle":"2026-04-15T07:16:00.352907Z","shell.execute_reply.started":"2026-04-15T07:16:00.337715Z","shell.execute_reply":"2026-04-15T07:16:00.352384Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"And the best part, this is not limited just to classification! So far Tensorflow looks much more simple than PyTorch. What happens if we try loading Vesuvius Dataset? \n\nSpoiler: The cell will run for some time before the kernel crashes!","metadata":{}},{"cell_type":"code","source":"import gc\ngc.collect()\ntf.keras.backend.clear_session() # Clears internal TF graph state","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-15T07:16:00.353719Z","iopub.execute_input":"2026-04-15T07:16:00.354091Z","iopub.status.idle":"2026-04-15T07:16:00.739725Z","shell.execute_reply.started":"2026-04-15T07:16:00.354062Z","shell.execute_reply":"2026-04-15T07:16:00.738969Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install imagecodecs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-15T07:16:00.740579Z","iopub.execute_input":"2026-04-15T07:16:00.740855Z","iopub.status.idle":"2026-04-15T07:16:06.173240Z","shell.execute_reply.started":"2026-04-15T07:16:00.740824Z","shell.execute_reply":"2026-04-15T07:16:06.172559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tifffile as tiff\nimport numpy as np\n\nclass VesuviusLoaderTF:\n    def __init__(self, fragment_path, target_size=(256, 256, 256)):\n        self.volume_dir = os.path.join(fragment_path, 'train_images')\n        self.labels_dir = os.path.join(fragment_path, 'train_labels')\n        self.target_size = target_size\n        \n        # Filter for .tif files\n        self.filenames = sorted([f for f in os.listdir(self.volume_dir) if f.endswith('.tif')])\n\n    def load_and_resize_volume(self, directory):\n        \"\"\"Loads and resizes each slice individually to keep size uniform\"\"\"\n        slices = []\n        # Fixing a size height and width. We convert 3D to 2D and restack\n        FIXED_2D_SIZE = (320, 320) \n        \n        for f in self.filenames:\n            img = tiff.imread(os.path.join(directory, f))\n            # Resize each 2D slice to a uniform size before stacking else you'll encounter errors!\n            img_resized = tf.image.resize(img[..., tf.newaxis], FIXED_2D_SIZE)\n            slices.append(tf.squeeze(img_resized, axis=-1).numpy())\n        \n        # Now stack will work because all arrays are (320, 320)\n        volume = np.stack(slices, axis=0).astype(np.float32) / 255.0\n        \n        # Continue with your 3D resizing logic to target_size \n        volume = tf.convert_to_tensor(volume)\n        return volume\n\n    def generator(self):\n        \"\"\"Yields a single 3D volume and its mask\"\"\"\n        # Keeping it similar to the pytorch version\n        volume = self.load_and_resize_volume(self.volume_dir)\n        mask = self.load_and_resize_volume(self.labels_dir)\n        yield volume, mask\n\n# Set the training pipeline\ndata_path = '/kaggle/input/competitions/vesuvius-challenge-surface-detection'\nloader = VesuviusLoaderTF(data_path)\n\n# Create the Dataset from the generator\ntrain_ds = tf.data.Dataset.from_generator(\n    loader.generator,\n    output_signature=(\n        tf.TensorSpec(shape=(256, 256, 256), dtype=tf.float32),\n        tf.TensorSpec(shape=(256, 256, 256), dtype=tf.float32)))\n\n# Apply optimizations\ntrain_ds = (\n    train_ds\n    .cache() \n    .repeat()          # Since len=1, repeat to keep training\n    .batch(1)          # Batch size 1\n    .prefetch(tf.data.AUTOTUNE)\n)\n\n# Test the shapes\nfor vol, mask in train_ds.take(1):\n    print(f\"Volume Shape: {vol.shape}\") # Expected: (1, 256, 256, 256)\n    print(f\"Mask Shape: {mask.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-15T07:16:06.174491Z","iopub.execute_input":"2026-04-15T07:16:06.174756Z","execution_failed":"2026-04-15T07:20:36.115Z"}},"outputs":[],"execution_count":null}]}