{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":6799,"databundleVersionId":4225553},{"sourceType":"datasetVersion","sourceId":2409323,"datasetId":1457432,"databundleVersionId":2451409}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"For using in real ImageNet =>\n\n- 1- change `DATA_PATH` to `/kaggle/input/competitions/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train`\n- 2- change `NUM_CLASSES_TO_USE` to `-1`\n- 3- Run","metadata":{}},{"cell_type":"code","source":"import os\nimport torch\nimport random\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.init as init\nfrom torchvision.datasets import ImageFolder\n\nfrom torchvision import datasets, transforms\nfrom torch.utils.data import DataLoader, random_split\nimport matplotlib.pyplot as plt\nimport numpy as np","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T10:59:09.538218Z","iopub.execute_input":"2026-03-23T10:59:09.538663Z","iopub.status.idle":"2026-03-23T10:59:17.999167Z","shell.execute_reply.started":"2026-03-23T10:59:09.538633Z","shell.execute_reply":"2026-03-23T10:59:17.998548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# Config\n# =====================\nDATA_PATH = \"/kaggle/input/datasets/arjunashok33/miniimagenet\"\nBATCH_SIZE = 128\nEPOCHS = 10\nIMAGE_SIZE = 227\nWORKERS = 4\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\ntorch.backends.cudnn.benchmark = True\n\nNUM_CLASSES_TO_USE = 10   # -1 => all","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T10:59:18.000367Z","iopub.execute_input":"2026-03-23T10:59:18.00096Z","iopub.status.idle":"2026-03-23T10:59:18.260317Z","shell.execute_reply.started":"2026-03-23T10:59:18.000934Z","shell.execute_reply":"2026-03-23T10:59:18.259498Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# Data\n# =====================\ntrain_transform = transforms.Compose([\n    transforms.Resize(256),\n    transforms.RandomCrop(227),\n    transforms.RandomHorizontalFlip(),\n    transforms.ToTensor(),\n    transforms.Normalize([0.485,0.456,0.406],[0.229,0.224,0.225])\n])\n\nval_transform = transforms.Compose([\n    transforms.Resize(256),\n    transforms.CenterCrop(IMAGE_SIZE),\n    transforms.ToTensor(),\n    transforms.Normalize([0.485,0.456,0.406],[0.229,0.224,0.225])\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T10:59:18.261875Z","iopub.execute_input":"2026-03-23T10:59:18.262263Z","iopub.status.idle":"2026-03-23T10:59:18.281036Z","shell.execute_reply.started":"2026-03-23T10:59:18.262217Z","shell.execute_reply":"2026-03-23T10:59:18.280327Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_classes = sorted(os.listdir(DATA_PATH))\nprint(len(all_classes))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T10:59:18.282641Z","iopub.execute_input":"2026-03-23T10:59:18.28341Z","iopub.status.idle":"2026-03-23T10:59:18.325571Z","shell.execute_reply.started":"2026-03-23T10:59:18.283373Z","shell.execute_reply":"2026-03-23T10:59:18.324909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if NUM_CLASSES_TO_USE != -1:\n    selected_classes = random.sample(all_classes, NUM_CLASSES_TO_USE)\nelse:\n    selected_classes = all_classes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T10:59:18.326509Z","iopub.execute_input":"2026-03-23T10:59:18.32681Z","iopub.status.idle":"2026-03-23T10:59:18.33033Z","shell.execute_reply.started":"2026-03-23T10:59:18.326786Z","shell.execute_reply":"2026-03-23T10:59:18.329789Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_classes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T10:59:18.331532Z","iopub.execute_input":"2026-03-23T10:59:18.331848Z","iopub.status.idle":"2026-03-23T10:59:18.346455Z","shell.execute_reply.started":"2026-03-23T10:59:18.331815Z","shell.execute_reply":"2026-03-23T10:59:18.345756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class_to_idx = {cls_name: i for i, cls_name in enumerate(selected_classes)}\nclass_to_idx","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T10:59:18.347369Z","iopub.execute_input":"2026-03-23T10:59:18.34804Z","iopub.status.idle":"2026-03-23T10:59:18.358651Z","shell.execute_reply.started":"2026-03-23T10:59:18.348003Z","shell.execute_reply":"2026-03-23T10:59:18.357945Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"subset_path = \"/kaggle/working/subset_data\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T10:59:18.359505Z","iopub.execute_input":"2026-03-23T10:59:18.359753Z","iopub.status.idle":"2026-03-23T10:59:18.369392Z","shell.execute_reply.started":"2026-03-23T10:59:18.359732Z","shell.execute_reply":"2026-03-23T10:59:18.368638Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if not os.path.exists(subset_path):\n    os.makedirs(subset_path)\n\n    for cls in selected_classes:\n        src = os.path.join(DATA_PATH, cls)\n        dst = os.path.join(subset_path, cls)\n        os.symlink(src, dst)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T10:59:18.370198Z","iopub.execute_input":"2026-03-23T10:59:18.370456Z","iopub.status.idle":"2026-03-23T10:59:18.382652Z","shell.execute_reply.started":"2026-03-23T10:59:18.370434Z","shell.execute_reply":"2026-03-23T10:59:18.381909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset = datasets.ImageFolder(subset_path, transform=train_transform)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T10:59:18.385092Z","iopub.execute_input":"2026-03-23T10:59:18.385375Z","iopub.status.idle":"2026-03-23T11:00:00.349529Z","shell.execute_reply.started":"2026-03-23T10:59:18.385352Z","shell.execute_reply":"2026-03-23T11:00:00.348662Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(len(dataset.classes))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T11:00:00.350712Z","iopub.execute_input":"2026-03-23T11:00:00.351407Z","iopub.status.idle":"2026-03-23T11:00:00.355593Z","shell.execute_reply.started":"2026-03-23T11:00:00.351379Z","shell.execute_reply":"2026-03-23T11:00:00.354773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_size = int(0.9 * len(dataset))\nval_size = len(dataset) - train_size\n\ntrain_data, val_data = random_split(dataset, [train_size, val_size])\n\nval_data.dataset.transform = val_transform","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T11:00:00.357242Z","iopub.execute_input":"2026-03-23T11:00:00.35817Z","iopub.status.idle":"2026-03-23T11:00:00.386338Z","shell.execute_reply.started":"2026-03-23T11:00:00.358127Z","shell.execute_reply":"2026-03-23T11:00:00.385518Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ntrain_loader = DataLoader(train_data, \n                          batch_size=BATCH_SIZE, \n                          shuffle=True, \n                          num_workers=WORKERS, \n                          pin_memory=True,\n                          persistent_workers=True,\n                          prefetch_factor=2)\nval_loader = DataLoader(val_data, \n                        batch_size=BATCH_SIZE, \n                        num_workers=WORKERS, \n                        pin_memory=True,\n                        persistent_workers=True,\n                        prefetch_factor=2)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T11:00:00.387439Z","iopub.execute_input":"2026-03-23T11:00:00.387805Z","iopub.status.idle":"2026-03-23T11:00:00.393173Z","shell.execute_reply.started":"2026-03-23T11:00:00.387781Z","shell.execute_reply":"2026-03-23T11:00:00.392447Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# Model (AlexNet)\n# =====================\nmodel = nn.Sequential(\n\n    nn.Conv2d(3, 96, kernel_size=11, stride=4),\n    nn.ReLU(),\n    nn.LocalResponseNorm(5),\n    nn.MaxPool2d(3, stride=2),\n\n    nn.Conv2d(96, 256, kernel_size=5, padding=2, groups=2),\n    nn.ReLU(),\n    nn.LocalResponseNorm(5),\n    nn.MaxPool2d(3, stride=2),\n\n    nn.Conv2d(256, 384, kernel_size=3, padding=1),\n    nn.ReLU(),\n\n    nn.Conv2d(384, 384, kernel_size=3, padding=1, groups=2),\n    nn.ReLU(),\n\n    nn.Conv2d(384, 256, kernel_size=3, padding=1, groups=2),\n    nn.ReLU(),\n    nn.MaxPool2d(3, stride=2),\n\n    nn.Flatten(),\n\n    nn.Linear(256*6*6, 4096),\n    nn.ReLU(),\n    nn.Dropout(0.5),\n\n    nn.Linear(4096, 4096),\n    nn.ReLU(),\n    nn.Dropout(0.5),\n\n    nn.Linear(4096, len(dataset.classes))\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T11:00:00.394209Z","iopub.execute_input":"2026-03-23T11:00:00.39456Z","iopub.status.idle":"2026-03-23T11:00:00.841469Z","shell.execute_reply.started":"2026-03-23T11:00:00.394524Z","shell.execute_reply":"2026-03-23T11:00:00.840833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# Multi-GPU \n# =====================\nif torch.cuda.device_count() > 1:\n    model = nn.DataParallel(model)\n\nmodel = model.to(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T11:00:00.842323Z","iopub.execute_input":"2026-03-23T11:00:00.842619Z","iopub.status.idle":"2026-03-23T11:00:01.220599Z","shell.execute_reply.started":"2026-03-23T11:00:00.842595Z","shell.execute_reply":"2026-03-23T11:00:01.219851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# Initialization\n# =====================\nfor layer in model.modules():\n    if isinstance(layer, nn.Conv2d) or isinstance(layer, nn.Linear):\n        init.normal_(layer.weight, 0, 0.01)\n        if layer.bias is not None:\n            nn.init.zeros_(layer.bias)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T11:00:01.221499Z","iopub.execute_input":"2026-03-23T11:00:01.221709Z","iopub.status.idle":"2026-03-23T11:00:01.25671Z","shell.execute_reply.started":"2026-03-23T11:00:01.221688Z","shell.execute_reply":"2026-03-23T11:00:01.256082Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# Training Setup\n# =====================\ncriterion = nn.CrossEntropyLoss()\n\noptimizer = optim.SGD(\n    model.parameters(),\n    lr=0.01,\n    momentum=0.9,\n    weight_decay=5e-4\n)\n\nscheduler = optim.lr_scheduler.StepLR(optimizer, step_size=10, gamma=0.1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T11:00:01.257577Z","iopub.execute_input":"2026-03-23T11:00:01.257944Z","iopub.status.idle":"2026-03-23T11:00:01.262684Z","shell.execute_reply.started":"2026-03-23T11:00:01.25792Z","shell.execute_reply":"2026-03-23T11:00:01.261994Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# Tracking\n# =====================\ntrain_losses = []\nval_losses = []","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T11:00:01.263749Z","iopub.execute_input":"2026-03-23T11:00:01.264055Z","iopub.status.idle":"2026-03-23T11:00:01.279183Z","shell.execute_reply.started":"2026-03-23T11:00:01.264021Z","shell.execute_reply":"2026-03-23T11:00:01.278276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# Training Loop\n# =====================\nfor epoch in range(EPOCHS):\n\n    # ---- Train ----\n    model.train()\n    train_loss = 0\n\n    for images, labels in train_loader:\n        images, labels = images.to(device), labels.to(device)\n\n        outputs = model(images)\n        loss = criterion(outputs, labels)\n\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n\n        train_loss += loss.item()\n\n    train_loss /= len(train_loader)\n\n    # ---- Validation ----\n    model.eval()\n    val_loss = 0\n\n    with torch.no_grad():\n        for images, labels in val_loader:\n            images, labels = images.to(device), labels.to(device)\n\n            outputs = model(images)\n            loss = criterion(outputs, labels)\n\n            val_loss += loss.item()\n\n    val_loss /= len(val_loader)\n\n    train_losses.append(train_loss)\n    val_losses.append(val_loss)\n\n    print(f\"Epoch {epoch+1} => train_loss: {train_loss:.4f}, val_loss: {val_loss:.4f}\")\n\n    scheduler.step()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T11:00:01.280265Z","iopub.execute_input":"2026-03-23T11:00:01.280645Z","iopub.status.idle":"2026-03-23T11:02:48.54835Z","shell.execute_reply.started":"2026-03-23T11:00:01.280621Z","shell.execute_reply":"2026-03-23T11:02:48.546556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# Plot Loss\n# =====================\nplt.figure()\nplt.plot(train_losses, label=\"Train Loss\")\nplt.plot(val_losses, label=\"Validation Loss\")\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Loss\")\nplt.title(\"Learning Curve\")\nplt.legend()\nplt.grid()\nplt.show()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-23T11:02:48.549945Z","iopub.status.idle":"2026-03-23T11:02:48.550393Z","shell.execute_reply.started":"2026-03-23T11:02:48.550161Z","shell.execute_reply":"2026-03-23T11:02:48.550191Z"}},"outputs":[],"execution_count":null}]}