{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":21154,"databundleVersionId":1243559}],"dockerImageVersionId":31286,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport re\nimport glob\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport io","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:51:39.848113Z","iopub.execute_input":"2026-02-23T09:51:39.848809Z","iopub.status.idle":"2026-02-23T09:51:40.169752Z","shell.execute_reply.started":"2026-02-23T09:51:39.848768Z","shell.execute_reply":"2026-02-23T09:51:40.169037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.transforms as transforms","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:51:40.171006Z","iopub.execute_input":"2026-02-23T09:51:40.171797Z","iopub.status.idle":"2026-02-23T09:51:42.980099Z","shell.execute_reply.started":"2026-02-23T09:51:40.171773Z","shell.execute_reply":"2026-02-23T09:51:42.979325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import timm\nfrom sklearn.metrics import f1_score\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:51:42.981049Z","iopub.execute_input":"2026-02-23T09:51:42.981446Z","iopub.status.idle":"2026-02-23T09:51:45.412042Z","shell.execute_reply.started":"2026-02-23T09:51:42.981421Z","shell.execute_reply":"2026-02-23T09:51:45.411268Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check device\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Device: {device}\")\nprint(f\"GPU: {torch.cuda.get_device_name(0) if torch.cuda.is_available() else 'None'}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:51:45.413019Z","iopub.execute_input":"2026-02-23T09:51:45.413463Z","iopub.status.idle":"2026-02-23T09:51:45.474755Z","shell.execute_reply.started":"2026-02-23T09:51:45.413430Z","shell.execute_reply":"2026-02-23T09:51:45.474148Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Paths\nTRAIN_PATH = \"/kaggle/input/tpu-getting-started/tfrecords-jpeg-512x512/train\"\nVAL_PATH   = \"/kaggle/input/tpu-getting-started/tfrecords-jpeg-512x512/val\"\nTEST_PATH  = \"/kaggle/input/tpu-getting-started/tfrecords-jpeg-512x512/test\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:51:45.476653Z","iopub.execute_input":"2026-02-23T09:51:45.477065Z","iopub.status.idle":"2026-02-23T09:51:45.480359Z","shell.execute_reply.started":"2026-02-23T09:51:45.477042Z","shell.execute_reply":"2026-02-23T09:51:45.479709Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# List files\ntrain_files = sorted(glob.glob(os.path.join(TRAIN_PATH, \"*.tfrec\")))\nval_files   = sorted(glob.glob(os.path.join(VAL_PATH,   \"*.tfrec\")))\ntest_files  = sorted(glob.glob(os.path.join(TEST_PATH,  \"*.tfrec\")))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:51:45.481188Z","iopub.execute_input":"2026-02-23T09:51:45.481555Z","iopub.status.idle":"2026-02-23T09:51:45.493454Z","shell.execute_reply.started":"2026-02-23T09:51:45.481511Z","shell.execute_reply":"2026-02-23T09:51:45.492819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Train files: {len(train_files)}\")\nprint(f\"Val files:   {len(val_files)}\")\nprint(f\"Test files:  {len(test_files)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:51:45.494266Z","iopub.execute_input":"2026-02-23T09:51:45.494541Z","iopub.status.idle":"2026-02-23T09:51:45.501386Z","shell.execute_reply.started":"2026-02-23T09:51:45.494513Z","shell.execute_reply":"2026-02-23T09:51:45.500700Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## We will use TF for parsing","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:51:45.502302Z","iopub.execute_input":"2026-02-23T09:51:45.502669Z","iopub.status.idle":"2026-02-23T09:51:54.620557Z","shell.execute_reply.started":"2026-02-23T09:51:45.502638Z","shell.execute_reply":"2026-02-23T09:51:54.619931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Disable GPU for TF since PyTorch is using it\ntf.config.set_visible_devices([], 'GPU')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:51:54.621390Z","iopub.execute_input":"2026-02-23T09:51:54.621931Z","iopub.status.idle":"2026-02-23T09:51:54.629068Z","shell.execute_reply.started":"2026-02-23T09:51:54.621890Z","shell.execute_reply":"2026-02-23T09:51:54.628387Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"FEATURE_DESCRIPTION = {\n    'image': tf.io.FixedLenFeature([], tf.string),\n    'class': tf.io.FixedLenFeature([], tf.int64),\n    'id':    tf.io.FixedLenFeature([], tf.string),\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:51:54.630035Z","iopub.execute_input":"2026-02-23T09:51:54.630335Z","iopub.status.idle":"2026-02-23T09:51:54.641467Z","shell.execute_reply.started":"2026-02-23T09:51:54.630310Z","shell.execute_reply":"2026-02-23T09:51:54.640836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def parse_tfrecord(example_proto):\n    parsed = tf.io.parse_single_example(example_proto, FEATURE_DESCRIPTION)\n    image = parsed['image'].numpy()   # raw JPEG bytes\n    label = int(parsed['class'].numpy())\n    img_id = parsed['id'].numpy().decode('utf-8')\n    return image, label, img_id","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:51:54.642312Z","iopub.execute_input":"2026-02-23T09:51:54.642595Z","iopub.status.idle":"2026-02-23T09:51:54.652825Z","shell.execute_reply.started":"2026-02-23T09:51:54.642564Z","shell.execute_reply":"2026-02-23T09:51:54.652095Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Peek at one record from train\nsample_dataset = tf.data.TFRecordDataset(train_files[0])\nfor raw in sample_dataset.take(1):\n    image_bytes, label, img_id = parse_tfrecord(raw)\n    img = Image.open(io.BytesIO(image_bytes))\n    print(f\"ID:    {img_id}\")\n    print(f\"Label: {label}\")\n    print(f\"Image size: {img.size}\")\n    print(f\"Image mode: {img.mode}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:51:54.653608Z","iopub.execute_input":"2026-02-23T09:51:54.653878Z","iopub.status.idle":"2026-02-23T09:51:54.727085Z","shell.execute_reply.started":"2026-02-23T09:51:54.653842Z","shell.execute_reply":"2026-02-23T09:51:54.726333Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load all records into memory","metadata":{}},{"cell_type":"code","source":"FEATURE_DESCRIPTION_TEST = {\n    'image': tf.io.FixedLenFeature([], tf.string),\n    'id':    tf.io.FixedLenFeature([], tf.string),\n} # there are no class field in test, iguess","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:51:54.728203Z","iopub.execute_input":"2026-02-23T09:51:54.728920Z","iopub.status.idle":"2026-02-23T09:51:54.732468Z","shell.execute_reply.started":"2026-02-23T09:51:54.728889Z","shell.execute_reply":"2026-02-23T09:51:54.731815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_tfrec_files(file_list, has_labels=True):\n    images, labels, ids = [], [], []\n    feature_desc = FEATURE_DESCRIPTION if has_labels else FEATURE_DESCRIPTION_TEST\n    for filepath in tqdm(file_list, desc=\"Loading files\"):\n        dataset = tf.data.TFRecordDataset(filepath)\n        for raw in dataset:\n            parsed = tf.io.parse_single_example(raw, feature_desc)\n            images.append(parsed['image'].numpy())\n            ids.append(parsed['id'].numpy().decode('utf-8'))\n            if has_labels:\n                labels.append(int(parsed['class'].numpy()))\n    return images, labels, ids\n\nprint('Loading train...')\ntrain_images, train_labels, train_ids = load_tfrec_files(train_files)\n\nprint('Loading val...')\nval_images, val_labels, val_ids = load_tfrec_files(val_files)\n\nprint('Loading test...')\ntest_images, test_labels, test_ids = load_tfrec_files(test_files, has_labels=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:51:54.735459Z","iopub.execute_input":"2026-02-23T09:51:54.735701Z","iopub.status.idle":"2026-02-23T09:52:14.733056Z","shell.execute_reply.started":"2026-02-23T09:51:54.735679Z","shell.execute_reply":"2026-02-23T09:52:14.732330Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"\\nTrain samples: {len(train_images)}\")\nprint(f\"Val samples:   {len(val_images)}\")\nprint(f\"Test samples:  {len(test_images)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:14.734067Z","iopub.execute_input":"2026-02-23T09:52:14.734360Z","iopub.status.idle":"2026-02-23T09:52:14.739436Z","shell.execute_reply.started":"2026-02-23T09:52:14.734335Z","shell.execute_reply":"2026-02-23T09:52:14.738781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import collections\ncounter = collections.Counter(train_labels)\nprint(f\"\\nNum unique classes in train: {len(counter)}\")\nprint(f\"Min samples per class: {min(counter.values())}\")\nprint(f\"Max samples per class: {max(counter.values())}\")\nprint(f\"Mean samples per class: {np.mean(list(counter.values())):.1f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:14.740399Z","iopub.execute_input":"2026-02-23T09:52:14.740715Z","iopub.status.idle":"2026-02-23T09:52:14.754008Z","shell.execute_reply.started":"2026-02-23T09:52:14.740684Z","shell.execute_reply":"2026-02-23T09:52:14.753318Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## PyTorch Dataset","metadata":{}},{"cell_type":"code","source":"import gc\ntorch.cuda.empty_cache()\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:14.754881Z","iopub.execute_input":"2026-02-23T09:52:14.755240Z","iopub.status.idle":"2026-02-23T09:52:15.080825Z","shell.execute_reply.started":"2026-02-23T09:52:14.755209Z","shell.execute_reply":"2026-02-23T09:52:15.080170Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"NUM_CLASSES = 104\nIMG_SIZE = 384","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:15.081742Z","iopub.execute_input":"2026-02-23T09:52:15.082120Z","iopub.status.idle":"2026-02-23T09:52:15.091882Z","shell.execute_reply.started":"2026-02-23T09:52:15.082086Z","shell.execute_reply":"2026-02-23T09:52:15.091149Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_transforms = transforms.Compose([\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomVerticalFlip(),\n    transforms.RandomRotation(30),\n    transforms.ColorJitter(brightness=0.2, contrast=0.2, saturation=0.2, hue=0.1),\n    transforms.RandomResizedCrop(IMG_SIZE, scale=(0.7, 1.0)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406],\n                         std=[0.229, 0.224, 0.225]),\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:15.092795Z","iopub.execute_input":"2026-02-23T09:52:15.093484Z","iopub.status.idle":"2026-02-23T09:52:15.104056Z","shell.execute_reply.started":"2026-02-23T09:52:15.093462Z","shell.execute_reply":"2026-02-23T09:52:15.103315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_transforms = transforms.Compose([\n    transforms.Resize((IMG_SIZE, IMG_SIZE)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406],\n                         std=[0.229, 0.224, 0.225]),\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:15.104962Z","iopub.execute_input":"2026-02-23T09:52:15.105325Z","iopub.status.idle":"2026-02-23T09:52:15.114605Z","shell.execute_reply.started":"2026-02-23T09:52:15.105295Z","shell.execute_reply":"2026-02-23T09:52:15.113890Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Dataset class","metadata":{}},{"cell_type":"code","source":"class FlowerDataset(Dataset):\n    def __init__(self, images, labels=None, ids=None, transform=None):\n        self.images = images\n        self.labels = labels\n        self.ids = ids\n        self.transform = transform\n    \n    def __len__(self):\n        return len(self.images)\n\n    def __getitem__(self, idx):\n        img = Image.open(io.BytesIO(self.images[idx])).convert(\"RGB\")\n        if self.transform:\n            img = self.transform(img)\n        if self.labels is not None:\n            return img, self.labels[idx]\n        return img, self.ids[idx]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:15.115468Z","iopub.execute_input":"2026-02-23T09:52:15.115719Z","iopub.status.idle":"2026-02-23T09:52:15.130882Z","shell.execute_reply.started":"2026-02-23T09:52:15.115700Z","shell.execute_reply":"2026-02-23T09:52:15.130090Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Weighted sampler to handle class imbalance\ndef make_weighted_sampler(labels):\n    counter = collections.Counter(labels)\n    class_weights = {cls: 1.0 / count for cls, count in counter.items()}\n    sample_weights = [class_weights[l] for l in labels]\n    return torch.utils.data.WeightedRandomSampler(\n        weights=sample_weights,\n        num_samples=len(sample_weights),\n        replacement=True\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:15.131735Z","iopub.execute_input":"2026-02-23T09:52:15.132020Z","iopub.status.idle":"2026-02-23T09:52:15.141822Z","shell.execute_reply.started":"2026-02-23T09:52:15.131991Z","shell.execute_reply":"2026-02-23T09:52:15.141271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Datasets\ntrain_dataset = FlowerDataset(train_images, train_labels, train_ids, transform=train_transforms)\nval_dataset   = FlowerDataset(val_images,   val_labels,   val_ids,   transform=val_transforms)\ntest_dataset  = FlowerDataset(test_images,  labels=None,  ids=test_ids, transform=val_transforms)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:15.142590Z","iopub.execute_input":"2026-02-23T09:52:15.143408Z","iopub.status.idle":"2026-02-23T09:52:15.153445Z","shell.execute_reply.started":"2026-02-23T09:52:15.143383Z","shell.execute_reply":"2026-02-23T09:52:15.152842Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Dataloaders\nBATCH_SIZE = 32","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:15.154123Z","iopub.execute_input":"2026-02-23T09:52:15.154352Z","iopub.status.idle":"2026-02-23T09:52:15.164137Z","shell.execute_reply.started":"2026-02-23T09:52:15.154321Z","shell.execute_reply":"2026-02-23T09:52:15.163571Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_loader = DataLoader(\n    train_dataset,\n    batch_size=BATCH_SIZE,\n    sampler=make_weighted_sampler(train_labels),\n    num_workers=2,\n    pin_memory=True\n)\n\nval_loader = DataLoader(\n    val_dataset,\n    batch_size=BATCH_SIZE,\n    shuffle=False,\n    num_workers=2,\n    pin_memory=True\n)\n\ntest_loader = DataLoader(\n    test_dataset,\n    batch_size=BATCH_SIZE,\n    shuffle=False,\n    num_workers=2,\n    pin_memory=True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:15.164940Z","iopub.execute_input":"2026-02-23T09:52:15.165274Z","iopub.status.idle":"2026-02-23T09:52:15.179719Z","shell.execute_reply.started":"2026-02-23T09:52:15.165221Z","shell.execute_reply":"2026-02-23T09:52:15.179125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Train batches: {len(train_loader)}\")\nprint(f\"Val batches:   {len(val_loader)}\")\nprint(f\"Test batches:  {len(test_loader)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:15.180511Z","iopub.execute_input":"2026-02-23T09:52:15.180814Z","iopub.status.idle":"2026-02-23T09:52:15.191817Z","shell.execute_reply.started":"2026-02-23T09:52:15.180794Z","shell.execute_reply":"2026-02-23T09:52:15.191222Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sanity check one batch\nimgs, lbls = next(iter(train_loader))\nprint(f\"\\nBatch image shape: {imgs.shape}\")\nprint(f\"Batch label shape: {lbls.shape}\")\nprint(f\"Label range: {lbls.min().item()} - {lbls.max().item()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:15.192566Z","iopub.execute_input":"2026-02-23T09:52:15.192803Z","iopub.status.idle":"2026-02-23T09:52:17.721110Z","shell.execute_reply.started":"2026-02-23T09:52:15.192773Z","shell.execute_reply":"2026-02-23T09:52:17.720409Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model setup","metadata":{}},{"cell_type":"code","source":"EPOCHS = 20\nLR = 1e-4\nNUM_CLASSES = 104","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:17.722267Z","iopub.execute_input":"2026-02-23T09:52:17.722518Z","iopub.status.idle":"2026-02-23T09:52:17.726373Z","shell.execute_reply.started":"2026-02-23T09:52:17.722483Z","shell.execute_reply":"2026-02-23T09:52:17.725674Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"try:\n    del model\nexcept:\n    pass\n\ntorch.cuda.empty_cache()\ngc.collect()\n\nprint(f\"GPU memory allocated: {torch.cuda.memory_allocated()/1024**3:.2f} GB\")\nprint(f\"GPU memory reserved:  {torch.cuda.memory_reserved()/1024**3:.2f} GB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:17.727356Z","iopub.execute_input":"2026-02-23T09:52:17.727602Z","iopub.status.idle":"2026-02-23T09:52:18.128834Z","shell.execute_reply.started":"2026-02-23T09:52:17.727578Z","shell.execute_reply":"2026-02-23T09:52:18.128080Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = timm.create_model('efficientnet_b5', pretrained=True, num_classes=NUM_CLASSES)\nmodel = model.to(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:18.129759Z","iopub.execute_input":"2026-02-23T09:52:18.130110Z","iopub.status.idle":"2026-02-23T09:52:18.986825Z","shell.execute_reply.started":"2026-02-23T09:52:18.130079Z","shell.execute_reply":"2026-02-23T09:52:18.986067Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"criterion = nn.CrossEntropyLoss(label_smoothing=0.1)\noptimizer = optim.AdamW(model.parameters(), lr=LR, weight_decay=1e-4)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:18.987925Z","iopub.execute_input":"2026-02-23T09:52:18.988201Z","iopub.status.idle":"2026-02-23T09:52:18.995021Z","shell.execute_reply.started":"2026-02-23T09:52:18.988162Z","shell.execute_reply":"2026-02-23T09:52:18.993986Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_scheduler(optimizer, num_warmup_steps, num_training_steps):\n    def lr_lambda(current_step):\n        if current_step < num_warmup_steps:\n            return float(current_step) / float(max(1, num_warmup_steps))\n        progress = float(current_step - num_warmup_steps) / float(max(1, num_training_steps - num_warmup_steps))\n        return max(0.0, 0.5 * (1.0 + np.cos(np.pi * progress)))\n    return torch.optim.lr_scheduler.LambdaLR(optimizer, lr_lambda)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:18.996113Z","iopub.execute_input":"2026-02-23T09:52:18.996416Z","iopub.status.idle":"2026-02-23T09:52:19.009162Z","shell.execute_reply.started":"2026-02-23T09:52:18.996393Z","shell.execute_reply":"2026-02-23T09:52:19.008451Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"total_steps = EPOCHS * len(train_loader)\nwarmup_steps = 1 * len(train_loader)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:19.010093Z","iopub.execute_input":"2026-02-23T09:52:19.010413Z","iopub.status.idle":"2026-02-23T09:52:19.019436Z","shell.execute_reply.started":"2026-02-23T09:52:19.010382Z","shell.execute_reply":"2026-02-23T09:52:19.018735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scheduler = get_scheduler(optimizer, warmup_steps, total_steps)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:19.020375Z","iopub.execute_input":"2026-02-23T09:52:19.021032Z","iopub.status.idle":"2026-02-23T09:52:19.030646Z","shell.execute_reply.started":"2026-02-23T09:52:19.020993Z","shell.execute_reply":"2026-02-23T09:52:19.030022Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count parameters\ntotal_params = sum(p.numel() for p in model.parameters() if p.requires_grad)\nprint(f\"Model: EfficientNet B5\")\nprint(f\"Trainable parameters: {total_params:,}\")\nprint(f\"Total epochs: {EPOCHS}\")\nprint(f\"Total training steps: {total_steps}\")\nprint(f\"Warmup steps: {warmup_steps}\")\nprint(f\"Criterion: CrossEntropyLoss (label_smoothing=0.1)\")\nprint(f\"Optimizer: AdamW (lr={LR}, weight_decay=1e-4)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:19.031433Z","iopub.execute_input":"2026-02-23T09:52:19.031725Z","iopub.status.idle":"2026-02-23T09:52:19.045325Z","shell.execute_reply.started":"2026-02-23T09:52:19.031695Z","shell.execute_reply":"2026-02-23T09:52:19.044606Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## LET US TRAIN!","metadata":{}},{"cell_type":"code","source":"from torch.cuda.amp import autocast, GradScaler\n\nscaler = GradScaler()\n\ndef train_epoch(model, loader, criterion, optimizer, scheduler):\n    model.train()\n    running_loss = 0.0\n    all_preds, all_labels = [], []\n\n    for imgs, labels in tqdm(loader, desc=\"Training\", leave=False):\n        imgs = imgs.to(device)\n        labels = labels.to(device)\n\n        optimizer.zero_grad()\n        \n        with autocast():\n            outputs = model(imgs)\n            loss = criterion(outputs, labels)\n        \n        scaler.scale(loss).backward()\n        scaler.unscale_(optimizer)\n        torch.nn.utils.clip_grad_norm_(model.parameters(), max_norm=1.0)\n        scaler.step(optimizer)\n        scaler.update()\n        scheduler.step()\n\n        running_loss += loss.item()\n        preds = outputs.argmax(dim=1).cpu().numpy()\n        all_preds.extend(preds)\n        all_labels.extend(labels.cpu().numpy())\n\n    avg_loss = running_loss / len(loader)\n    f1 = f1_score(all_labels, all_preds, average='macro', zero_division=0)\n    return avg_loss, f1\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:19.046220Z","iopub.execute_input":"2026-02-23T09:52:19.046986Z","iopub.status.idle":"2026-02-23T09:52:19.057452Z","shell.execute_reply.started":"2026-02-23T09:52:19.046964Z","shell.execute_reply":"2026-02-23T09:52:19.056827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def val_epoch(model, loader, criterion):\n    model.eval()\n    running_loss = 0.0\n    all_preds, all_labels = [], []\n\n    with torch.no_grad():\n        for imgs, labels in tqdm(loader, desc=\"Validating\", leave=False):\n            imgs = imgs.to(device)\n            labels = labels.to(device)\n\n            with autocast():\n                outputs = model(imgs)\n                loss = criterion(outputs, labels)\n\n            running_loss += loss.item()\n            preds = outputs.argmax(dim=1).cpu().numpy()\n            all_preds.extend(preds)\n            all_labels.extend(labels.cpu().numpy())\n\n    avg_loss = running_loss / len(loader)\n    f1 = f1_score(all_labels, all_preds, average='macro', zero_division=0)\n    return avg_loss, f1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:19.058144Z","iopub.execute_input":"2026-02-23T09:52:19.058390Z","iopub.status.idle":"2026-02-23T09:52:19.070837Z","shell.execute_reply.started":"2026-02-23T09:52:19.058368Z","shell.execute_reply":"2026-02-23T09:52:19.070139Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"torch.cuda.empty_cache()\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:19.071796Z","iopub.execute_input":"2026-02-23T09:52:19.072605Z","iopub.status.idle":"2026-02-23T09:52:19.407199Z","shell.execute_reply.started":"2026-02-23T09:52:19.072572Z","shell.execute_reply":"2026-02-23T09:52:19.406420Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Training loop\nbest_val_f1 = 0.0\nbest_model_path = \"/kaggle/working/best_model.pth\"\nhistory = []\n\nprint(f\"Starting training for {EPOCHS} epochs...\\n\")\n\nfor epoch in range(EPOCHS):\n    train_loss, train_f1 = train_epoch(model, train_loader, criterion, optimizer, scheduler)\n    val_loss, val_f1     = val_epoch(model, val_loader, criterion)\n\n    history.append({\n        'epoch': epoch + 1,\n        'train_loss': train_loss,\n        'train_f1': train_f1,\n        'val_loss': val_loss,\n        'val_f1': val_f1,\n        'lr': scheduler.get_last_lr()[0]\n    })\n\n    if val_f1 > best_val_f1:\n        best_val_f1 = val_f1\n        torch.save(model.state_dict(), best_model_path)\n        print(f\"Epoch {epoch+1:02d}/{EPOCHS} | \"\n              f\"Train Loss: {train_loss:.4f} | Train F1: {train_f1:.4f} | \"\n              f\"Val Loss: {val_loss:.4f} | Val F1: {val_f1:.4f} | \"\n              f\"LR: {scheduler.get_last_lr()[0]:.6f} (saved)\")\n    else:\n        print(f\"Epoch {epoch+1:02d}/{EPOCHS} | \"\n              f\"Train Loss: {train_loss:.4f} | Train F1: {train_f1:.4f} | \"\n              f\"Val Loss: {val_loss:.4f} | Val F1: {val_f1:.4f} | \"\n              f\"LR: {scheduler.get_last_lr()[0]:.6f}\")\n\nprint(f\"\\nBest Val F1: {best_val_f1:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:52:19.408289Z","iopub.execute_input":"2026-02-23T09:52:19.408606Z","iopub.status.idle":"2026-02-23T12:05:21.773326Z","shell.execute_reply.started":"2026-02-23T09:52:19.408564Z","shell.execute_reply":"2026-02-23T12:05:21.772532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict(model, loader):\n    model.eval()\n    all_preds, all_ids = [], []\n\n    with torch.no_grad():\n        for imgs, ids in tqdm(loader, desc=\"Predicting\"):\n            imgs = imgs.to(device)\n            with torch.amp.autocast('cuda'):\n                outputs = model(imgs)\n            preds = outputs.argmax(dim=1).cpu().numpy()\n            all_preds.extend(preds)\n            all_ids.extend(ids)\n\n    return all_ids, all_preds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T12:05:21.774779Z","iopub.execute_input":"2026-02-23T12:05:21.775076Z","iopub.status.idle":"2026-02-23T12:05:21.780715Z","shell.execute_reply.started":"2026-02-23T12:05:21.775046Z","shell.execute_reply":"2026-02-23T12:05:21.780061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.load_state_dict(torch.load(best_model_path))\nprint(\"Loaded best model\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T12:05:21.785304Z","iopub.execute_input":"2026-02-23T12:05:21.785551Z","iopub.status.idle":"2026-02-23T12:05:22.073064Z","shell.execute_reply.started":"2026-02-23T12:05:21.785525Z","shell.execute_reply":"2026-02-23T12:05:22.072408Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_ids, all_preds = predict(model, test_loader)\n\nsubmission = pd.DataFrame({'id': all_ids, 'label': all_preds})\nsubmission.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T12:05:22.073914Z","iopub.execute_input":"2026-02-23T12:05:22.074195Z","iopub.status.idle":"2026-02-23T12:06:14.123202Z","shell.execute_reply.started":"2026-02-23T12:05:22.074159Z","shell.execute_reply":"2026-02-23T12:06:14.122511Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Submission shape: {submission.shape}\")\nprint(f\"Unique predicted classes: {submission['label'].nunique()}\")\nprint(submission.head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T12:06:14.124424Z","iopub.execute_input":"2026-02-23T12:06:14.124706Z","iopub.status.idle":"2026-02-23T12:06:14.143181Z","shell.execute_reply.started":"2026-02-23T12:06:14.124678Z","shell.execute_reply":"2026-02-23T12:06:14.142483Z"}},"outputs":[],"execution_count":null}]}