{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":6799,"databundleVersionId":4225553,"isSourceIdPinned":false,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader, Dataset\nfrom torchvision import datasets\nfrom torchvision.transforms import v2\nfrom torchvision.io import read_image\nfrom torch.optim import SGD\nimport numpy as np\nfrom tqdm import tqdm\nfrom torch.optim.lr_scheduler import StepLR\nimport os\nimport glob\nimport pandas as pd","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-16T13:41:28.821988Z","iopub.execute_input":"2025-07-16T13:41:28.822136Z","iopub.status.idle":"2025-07-16T13:41:36.239235Z","shell.execute_reply.started":"2025-07-16T13:41:28.822121Z","shell.execute_reply":"2025-07-16T13:41:36.23843Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Model code\ndef init_weights(module):\n    if isinstance(module, nn.Linear) or isinstance(module, nn.Conv2d):\n        nn.init.xavier_uniform_(module.weight)\n        if module.bias is not None:\n            nn.init.zeros_(module.bias)\n\nclass AlexNet(nn.Module):\n    def __init__(self, num_classes=1000):\n        super().__init__()\n        self.AlexNet = nn.Sequential(\n            nn.Conv2d(3, 96, kernel_size=11, stride=4), nn.ReLU(),\n            nn.MaxPool2d(kernel_size=3, stride=2),\n            nn.LocalResponseNorm(size=5),\n            nn.Conv2d(96, 256, kernel_size=5, padding=2), nn.ReLU(),\n            nn.MaxPool2d(kernel_size=3, stride=2),\n            nn.LocalResponseNorm(size=5),\n            nn.Conv2d(256, 384, kernel_size=3, padding=1), nn.ReLU(),\n            nn.Conv2d(384, 384, kernel_size=3, padding=1), nn.ReLU(),\n            nn.Conv2d(384, 256, kernel_size=3, padding=1), nn.ReLU(),\n            nn.MaxPool2d(kernel_size=3, stride=2), nn.Flatten(),\n            nn.Linear(256*5*5, 4096), nn.ReLU(), nn.Dropout(p=0.5),\n            nn.Linear(4096, 4096), nn.ReLU(), nn.Dropout(p=0.5),\n            nn.Linear(4096, num_classes)\n        )\n\n    def forward(self, X):\n        return self.AlexNet(X)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-11T12:23:14.759101Z","iopub.execute_input":"2025-07-11T12:23:14.759355Z","iopub.status.idle":"2025-07-11T12:23:14.766616Z","shell.execute_reply.started":"2025-07-11T12:23:14.759339Z","shell.execute_reply":"2025-07-11T12:23:14.76598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Conversion hashmaps\nsynsetToClass = {}\nclassToSynset = {}\nclassToDescription = {}\ni = 0\nwith open(\"/kaggle/input/imagenet-object-localization-challenge/LOC_synset_mapping.txt\",\"r\") as file:\n    for idx, line in enumerate(file):\n        parts = line.split(\" \",1)\n        synset = parts[0]\n        synsetToClass[synset] = i\n        classToSynset[i] = synset\n        classToDescription[i] = parts[1].strip() if len(parts) > 1 else ''\n        i += 1\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-11T12:23:14.768704Z","iopub.execute_input":"2025-07-11T12:23:14.769012Z","iopub.status.idle":"2025-07-11T12:23:14.812658Z","shell.execute_reply.started":"2025-07-11T12:23:14.768969Z","shell.execute_reply":"2025-07-11T12:23:14.812011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"os.makedirs(\"labels/train\", exist_ok = True)\nos.makedirs(\"labels/validation\", exist_ok = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-11T12:23:14.81323Z","iopub.execute_input":"2025-07-11T12:23:14.813381Z","iopub.status.idle":"2025-07-11T12:23:14.817042Z","shell.execute_reply.started":"2025-07-11T12:23:14.813368Z","shell.execute_reply":"2025-07-11T12:23:14.816267Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Creating a CSV file for the training data\ntrain_dir = glob.glob(\"/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train/*\")\ni = 0\nfor folder_name in tqdm(train_dir):\n    i += 1\n    synset = os.path.basename(folder_name)\n    class_no = synsetToClass[synset]\n\n    img_files = glob.glob(os.path.join(f\"/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train/{synset}/\", \"*.JPEG\"))\n    img_files = [os.path.basename(img) for img in img_files]\n\n    with open(f'labels/train/train_labels.csv', 'a') as f:\n        if i == 1: \n            f.write(\"image, class \\n\")\n        for img in img_files:\n            f.write(f'{img}, {class_no} \\n')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-11T12:23:14.817747Z","iopub.execute_input":"2025-07-11T12:23:14.818418Z","iopub.status.idle":"2025-07-11T12:23:48.692637Z","shell.execute_reply.started":"2025-07-11T12:23:14.818398Z","shell.execute_reply":"2025-07-11T12:23:48.691903Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/imagenet-object-localization-challenge/LOC_val_solution.csv')\ndf.sort_values(by='ImageId', inplace=True)\n\nj = 0\nfor i, row in tqdm(df.iterrows()):\n    j += 1\n    parts = row['PredictionString'].split(' ', 1)\n    img = row['ImageId'] + '.JPEG'\n    class_no = synsetToClass[parts[0]]\n    with open(f'labels/validation/val_labels.csv', 'a') as f:\n        f.write(f'{img}, {class_no} \\n')\n\nprint(f'Total number of valuation images: {j}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-11T12:23:48.693594Z","iopub.execute_input":"2025-07-11T12:23:48.693872Z","iopub.status.idle":"2025-07-11T12:23:52.726259Z","shell.execute_reply.started":"2025-07-11T12:23:48.693846Z","shell.execute_reply":"2025-07-11T12:23:52.725574Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ImageNetDataSet(Dataset):\n    def __init__(self,directory,annotations_file,transform = None, target_transform = None, train = True):\n        self.directory = directory\n        self.annotations = pd.read_csv(annotations_file)\n        self.transform = transform\n        self.target_transform = target_transform\n        self.train = train\n\n    def __len__(self):\n        return len(self.annotations)\n\n    def __getitem__(self,idx):\n        image_file = self.annotations.iloc[idx,0]\n        class_no = int(self.annotations.iloc[idx,1])\n\n        if self.train:\n            folder_path = os.path.join(self.directory,classToSynset[class_no])\n            image_path = os.path.join(folder_path,image_file)\n        else:\n            image_path = os.path.join(self.directory, image_file)\n\n        image = read_image(image_path)\n        if image.shape[0] == 1:\n            image = image.expand(3,-1,-1)\n\n        if image.shape[0] == 4:\n            image = image[:3,:,:]\n\n        if self.transform:\n            image = self.transform(image)\n\n        if self.target_transform:\n            class_no = self.target_transform(class_no)\n\n        return image, class_no\n            ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-11T12:23:52.726899Z","iopub.execute_input":"2025-07-11T12:23:52.727145Z","iopub.status.idle":"2025-07-11T12:23:52.733258Z","shell.execute_reply.started":"2025-07-11T12:23:52.727128Z","shell.execute_reply":"2025-07-11T12:23:52.732663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transform = v2.Compose([\n    v2.Resize(256),\n    v2.RandomCrop((224,224)),\n    v2.RandomHorizontalFlip(p = 0.5),\n    v2.ToImage(),\n    v2.ToDtype(torch.float32, scale = True),\n    v2.Normalize(mean = [0.485, 0.456, 0.406],  std = [0.229, 0.224, 0.225])\n])\n\ntrain_directory = \"/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train\"\ntrain_annotations = \"/kaggle/working/labels/train/train_labels.csv\"\n\ntrain_set = ImageNetDataSet(\n    directory = train_directory,\n    annotations_file = train_annotations,\n    transform = transform,\n    train = True\n)\n\nvalidation_directory = \"/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/val\"\nvalidation_annotations = \"/kaggle/working/labels/validation/val_labels.csv\"\n\nval_set = ImageNetDataSet(\n    directory = validation_directory,\n    annotations_file = validation_annotations,\n    transform = transform,\n    train = False\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-11T12:23:52.734021Z","iopub.execute_input":"2025-07-11T12:23:52.734252Z","iopub.status.idle":"2025-07-11T12:23:53.566836Z","shell.execute_reply.started":"2025-07-11T12:23:52.734229Z","shell.execute_reply":"2025-07-11T12:23:53.566136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataloader = DataLoader(train_set, batch_size=64, shuffle=True)\nval_dataloader = DataLoader(val_set, batch_size=64, shuffle=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-11T12:23:53.569138Z","iopub.execute_input":"2025-07-11T12:23:53.569384Z","iopub.status.idle":"2025-07-11T12:23:53.573346Z","shell.execute_reply.started":"2025-07-11T12:23:53.569366Z","shell.execute_reply":"2025-07-11T12:23:53.572725Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\ndevice","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-11T12:23:53.574046Z","iopub.execute_input":"2025-07-11T12:23:53.574286Z","iopub.status.idle":"2025-07-11T12:23:53.662057Z","shell.execute_reply.started":"2025-07-11T12:23:53.574264Z","shell.execute_reply":"2025-07-11T12:23:53.661359Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = AlexNet().to(device)\nif torch.cuda.device_count() > 1:\n    print(f\"Using {torch.cuda.device_count()} GPUs\")\n    model = nn.DataParallel(model)\n\nmodel.to(device)\nmodel.apply(init_weights)\noptimizer = SGD(model.parameters(), lr = 0.01, momentum = 0.9, weight_decay = 0.0005)\nloss_fn = nn.CrossEntropyLoss()\nscheduler = StepLR(\n    optimizer,\n    step_size=25,\n    gamma=0.1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-11T12:23:53.662864Z","iopub.execute_input":"2025-07-11T12:23:53.663405Z","iopub.status.idle":"2025-07-11T12:23:54.365041Z","shell.execute_reply.started":"2025-07-11T12:23:53.663383Z","shell.execute_reply":"2025-07-11T12:23:54.364308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train(model,train_dataloader,optimizer,loss_fn):\n    size = len(train_dataloader.dataset)\n    model.train()\n    total_loss = 0\n    for batch, (X, y) in tqdm(enumerate(train_dataloader),total = len(train_dataloader)):\n        X, y = X.to(device), y.to(device)\n\n        pred = model(X)\n        loss = loss_fn(pred, y)\n        total_loss += loss.item()\n\n        loss.backward()\n        optimizer.step()\n        optimizer.zero_grad()\n\n    scheduler.step()\n    avg_train_loss = total_loss / len(train_dataloader)\n    return avg_train_loss\n\ndef test(dataloader, model, loss_fn):\n    size = len(dataloader.dataset)\n    num_batches = len(dataloader)\n    model.eval()\n\n    test_loss, correct = 0, 0\n    with torch.no_grad():\n        for X, y in dataloader:\n            X, y = X.to(device), y.to(device)\n            pred = model(X)\n            test_loss += loss_fn(pred, y).item()\n            correct += (pred.argmax(1) == y).type(torch.float).sum().item()\n    avg_test_loss = test_loss / num_batches\n    accuracy = correct / size\n    return accuracy, avg_test_loss","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-11T12:23:54.365826Z","iopub.execute_input":"2025-07-11T12:23:54.366046Z","iopub.status.idle":"2025-07-11T12:23:54.372358Z","shell.execute_reply.started":"2025-07-11T12:23:54.366021Z","shell.execute_reply":"2025-07-11T12:23:54.371728Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"epochs = 1\nfor t in range(epochs):\n    avg_train_loss = train(model, train_dataloader,optimizer,loss_fn)\n    if t == 0 or (t + 1) % 5 == 0 or t == epochs - 1:  # Print every 5 epochs or the last epoch\n        accuracy, avg_test_loss = test(val_dataloader, model, loss_fn)\n        print(f\"Epoch {t+1}\\n-------------------------------\")\n        print(f\"Train Avg Loss: {avg_train_loss:>8f}\")\n        print(f\"Test Accuracy: {(100 * accuracy):>0.1f}%, Test Avg Loss: {avg_test_loss:>8f} \\n\")\nprint(\"Done!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-11T12:23:54.372967Z","iopub.execute_input":"2025-07-11T12:23:54.373184Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}