{"metadata":{"colab":{"provenance":[]},"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":6799,"databundleVersionId":4225553,"sourceType":"competition"}],"dockerImageVersionId":30733,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2024-06-07T22:01:54.222697Z","iopub.execute_input":"2024-06-07T22:01:54.223132Z","iopub.status.idle":"2024-06-07T22:01:54.229173Z","shell.execute_reply.started":"2024-06-07T22:01:54.2231Z","shell.execute_reply":"2024-06-07T22:01:54.228077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport torch\nimport torchvision as tv\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optims\nimport torchvision.transforms as transforms\nfrom PIL import Image\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2024-06-07T22:01:56.920952Z","iopub.execute_input":"2024-06-07T22:01:56.921357Z","iopub.status.idle":"2024-06-07T22:02:02.341536Z","shell.execute_reply.started":"2024-06-07T22:01:56.921325Z","shell.execute_reply":"2024-06-07T22:02:02.340601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resize_image(src, size=(128, 128), bgc=\"white\"):\n    src.thumbnail(size, Image.ANTIALIAS)\n    \n    new_image = Image.new(\"RGB\", size, bgc)\n    \n    new_image.paste(src, (int((size[0]-src.size[0]) / 2)), int((size[1] - src.size[1]) / 2))\n    \n    return new_image","metadata":{"execution":{"iopub.status.busy":"2024-06-07T22:02:04.22196Z","iopub.execute_input":"2024-06-07T22:02:04.223131Z","iopub.status.idle":"2024-06-07T22:02:04.229843Z","shell.execute_reply.started":"2024-06-07T22:02:04.22309Z","shell.execute_reply":"2024-06-07T22:02:04.228655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transform = transforms.Compose([\n    transforms.Resize([256, 256]),\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomVerticalFlip(),\n    transforms.ColorJitter(brightness=0.5, contrast=0),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.5, 0.5, 0.5], std=[0.5, 0.5, 0.5])\n])","metadata":{"execution":{"iopub.status.busy":"2024-06-07T22:02:05.505941Z","iopub.execute_input":"2024-06-07T22:02:05.50634Z","iopub.status.idle":"2024-06-07T22:02:05.513367Z","shell.execute_reply.started":"2024-06-07T22:02:05.506306Z","shell.execute_reply":"2024-06-07T22:02:05.512067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load(path):\n    dataset = tv.datasets.ImageFolder(root=path,transform = transform)\n    \n    train_size = int(0.7 * len(dataset))\n    test_size = len(dataset) - train_size\n    \n    train_set, test_set = torch.utils.data.random_split(dataset, [train_size, test_size])\n    \n    train_loader = torch.utils.data.DataLoader(\n        train_set,\n        batch_size=50,\n        num_workers=0,\n        shuffle=False\n    )\n        valid_loader = torch.utils.data.DataLoader(\n        train_set,\n        batch_size=50,\n        num_workers=0,\n        shuffle=False\n    )\n    \n    test_loader = torch.utils.data.DataLoader(\n        test_set,\n        batch_size=50,\n        num_workers=0,\n        shuffle=False\n    )\n\n    return train_loader, test_loader\n\ntest_loader, train_loader, valid_loader = load('/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC')\nprint(f\"train loaders: {train_loader}\")","metadata":{"execution":{"iopub.status.busy":"2024-06-07T22:02:40.561111Z","iopub.execute_input":"2024-06-07T22:02:40.562038Z","iopub.status.idle":"2024-06-07T22:03:01.03644Z","shell.execute_reply.started":"2024-06-07T22:02:40.561994Z","shell.execute_reply":"2024-06-07T22:03:01.034806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(device)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nmodel = tv.models.resnet50(weights=tv.models.ResNet50_Weights.DEFAULT)\n\nmodel = model.to(device)\n\ncriterion = torch.nn.CrossEntropyLoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=0.05)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss = []\n\nfor epoch, (inputs, labels) in enumerate(train_loader):\n    print(f'Epoch {epoch+1}')  \n          \n    # Move input and label tensors to the device\n    inputs = inputs.to(device)\n    labels = labels.to(device)\n\n    # Zero out the optimizer\n    optimizer.zero_grad()\n\n    # Forward pass\n    outputs = model(inputs)\n    print(f'Outputs: {outputs}')\n    loss.append(criterion(outputs, labels))\n\n    # Backward pass\n    loss[epoch].backward()\n    optimizer.step()\n\n    # Print the loss for every epoch\n    print(f'Loss: {loss[epoch].item():.4f}')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torchvision.transforms as transforms\nimport torchvision.datasets as datasets\nfrom PIL import Image\nimport torch.nn.init as init\nimport torch.optim as optim\nimport os","metadata":{"id":"DiotJrQTjgua"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the data augmentation transformations\ndata_transforms = transforms.Compose([\n    # Randomly resize and crop the image to 224x224\n    transforms.RandomResizedCrop(224),\n    # Randomly flip the image horizontally\n    transforms.RandomHorizontalFlip(),\n    # Randomly adjust brightness, contrast, saturation, and hue\n    transforms.ColorJitter(brightness=0.2, contrast=0.2, saturation=0.2, hue=0.1),\n    # Convert the image to a PyTorch tensor\n    transforms.ToTensor(),\n])\n\n# Define paths to save the model\nsave_dir = \"./models\"\nos.makedirs(save_dir, exist_ok=True)\n\n# Define data loaders for training and validation sets\ntrain_dataset = datasets.ImageNet(root=\"path/to/ImageNet/train\", split='train', transform=data_transforms)\nval_dataset = datasets.ImageNet(root=\"path/to/ImageNet/val\", split='val', transform=data_transforms)\n\ntrain_loader = torch.utils.data.DataLoader(train_dataset, batch_size=64, shuffle=True, num_workers=4)\nval_loader = torch.utils.data.DataLoader(val_dataset, batch_size=64, shuffle=False, num_workers=4)\n","metadata":{"id":"c5xNR0iLjUCw"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class AlexNet(nn.Module):\n    def __init__(self, num_classes=1000):\n        super(AlexNet, self).__init__()\n        self.features = nn.Sequential(\n            nn.Conv2d(3, 64, kernel_size=11, stride=4, padding=2),\n            nn.ReLU(inplace=True),\n            nn.LocalResponseNorm(size=5, alpha=0.0001, beta=0.75, k=2),  # LRN after conv1\n            nn.MaxPool2d(kernel_size=3, stride=2),\n\n            nn.Conv2d(64, 192, kernel_size=5, padding=2),\n            nn.ReLU(inplace=True),\n            nn.LocalResponseNorm(size=5, alpha=0.0001, beta=0.75, k=2),  # LRN after conv2\n            nn.MaxPool2d(kernel_size=3, stride=2),\n\n            nn.Conv2d(192, 384, kernel_size=3, padding=1),\n            nn.ReLU(inplace=True),\n\n            nn.Conv2d(384, 256, kernel_size=3, padding=1),\n            nn.ReLU(inplace=True),\n\n            nn.Conv2d(256, 256, kernel_size=3, padding=1),\n            nn.ReLU(inplace=True),\n            nn.MaxPool2d(kernel_size=3, stride=2),\n        )\n\n        self.avgpool = nn.AdaptiveAvgPool2d((6, 6))\n\n        self.classifier = nn.Sequential(\n            nn.Dropout(),\n            nn.Linear(256 * 6 * 6, 4096),\n            nn.ReLU(inplace=True),\n            nn.Dropout(),\n            nn.Linear(4096, 4096),\n            nn.ReLU(inplace=True),\n            nn.Linear(4096, num_classes),\n        )\n\n    def forward(self, x):\n        x = self.features(x)\n        x = torch.flatten(x, 1)\n        x = self.classifier(x)\n        return x\n\n    def _initialize_weights(self):\n        for m in self.modules():\n            if isinstance(m, nn.Conv2d) or isinstance(m, nn.Linear):\n                init.normal_(m.weight, 0, 0.01)\n                if m.bias is not None:\n                    init.constant_(m.bias, 1 if isinstance(m, nn.Conv2d) and m in [self.features[1], self.features[4], self.features[7]] else 0)\n\n","metadata":{"id":"35MdKDInjZCq"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Instantiate the model and initialize weights\nmodel = AlexNet(num_classes=1000)\nmodel._initialize_weights()\n\n# Define loss function and optimizer\ncriterion = nn.CrossEntropyLoss()\noptimizer = optim.SGD(model.parameters(), lr=0.01, momentum=0.9)\n\n# Move model to GPU if available\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel.to(device)\n\n# Training loop\nnum_epochs = 5\ntotal_step = len(train_loader)\n\nfor epoch in range(num_epochs):\n    model.train()\n    running_loss = 0.0\n    correct_train = 0\n    total_train = 0\n\n    for i, (images, labels) in enumerate(train_loader):\n        # Move tensors to the configured device\n        images = images.to(device)\n        labels = labels.to(device)\n\n        # Forward pass\n        outputs = model(images)\n        loss = criterion(outputs, labels)\n\n        # Backward and optimize\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n\n        running_loss += loss.item()\n        _, predicted = torch.max(outputs.data, 1)\n        total_train += labels.size(0)\n        correct_train += (predicted == labels).sum().item()\n\n    train_loss = running_loss / total_step\n    train_acc = correct_train / total_train\n\n    # Validation\n    model.eval()\n    val_loss = 0.0\n    correct_val = 0\n    total_val = 0\n    with torch.no_grad():\n        for images, labels in val_loader:\n            images = images.to(device)\n            labels = labels.to(device)\n            outputs = model(images)\n            loss = criterion(outputs, labels)\n            val_loss += loss.item()\n            _, predicted = torch.max(outputs.data, 1)\n            total_val += labels.size(0)\n            correct_val += (predicted == labels).sum().item()\n\n    val_loss = val_loss / len(val_loader)\n    val_acc = correct_val / total_val\n\n    print('Epoch [{}/{}], Train Loss: {:.4f}, Train Acc: {:.4f}, Val Loss: {:.4f}, Val Acc: {:.4f}'\n          .format(epoch+1, num_epochs, train_loss, train_acc, val_loss, val_acc))\n\n    # Save the model after each epoch\n    save_path = os.path.join(save_dir, f\"alexnet_epoch_{epoch + 1}.pt\")\n    torch.save(model.state_dict(), save_path)\n    print(f\"Model saved at: {save_path}\")","metadata":{"id":"PSDZgvHxjblX"},"execution_count":null,"outputs":[]}]}