{"metadata":{"colab":{"provenance":[]},"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":6799,"databundleVersionId":4225553,"sourceType":"competition"}],"dockerImageVersionId":30733,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2024-06-07T22:01:54.222697Z","iopub.execute_input":"2024-06-07T22:01:54.223132Z","iopub.status.idle":"2024-06-07T22:01:54.229173Z","shell.execute_reply.started":"2024-06-07T22:01:54.2231Z","shell.execute_reply":"2024-06-07T22:01:54.228077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport torch\nimport torchvision as tv\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nimport torchvision.transforms as transforms\n\nfrom PIL import Image\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nfrom glob import glob\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport random\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Device configuration\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')","metadata":{"execution":{"iopub.status.busy":"2024-06-08T18:29:41.056932Z","iopub.execute_input":"2024-06-08T18:29:41.057315Z","iopub.status.idle":"2024-06-08T18:29:41.066016Z","shell.execute_reply.started":"2024-06-08T18:29:41.05728Z","shell.execute_reply":"2024-06-08T18:29:41.065258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tqdm\n!pip install seaborn","metadata":{"execution":{"iopub.status.busy":"2024-06-08T14:08:17.27233Z","iopub.execute_input":"2024-06-08T14:08:17.272982Z","iopub.status.idle":"2024-06-08T14:08:42.27241Z","shell.execute_reply.started":"2024-06-08T14:08:17.272952Z","shell.execute_reply":"2024-06-08T14:08:42.271326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Adding Code","metadata":{}},{"cell_type":"markdown","source":"***Structure validation data well***","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom shutil import move\nimport xml.etree.ElementTree as ET\n\n# Path constants for Kaggle environment\nIMAGE_PATH_VALID = '/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/val/'\nANNOTATION_PATH = '/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Annotations/CLS-LOC/val/'\nIMAGE_PATH_TRAIN = '/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train/'\nSAVE_PATH = '/kaggle/working/'\n\n# Get the names of the validation dataset.\nimage_names_valid = os.listdir(IMAGE_PATH_VALID)\n\n# An empty list for the validation dataset labels.\nimage_labels_vald = []\nfor i in image_names_valid:\n    # Passing the path of the xml document to enable the parsing process\n    tree = ET.parse(os.path.join(ANNOTATION_PATH, i[:-5] + '.xml'))\n    # getting the parent tag of the xml document\n    root = tree.getroot()\n    image_labels_vald.append(root[5][0].text)\n\nvalidation_list = {\"Image_Name\": image_names_valid, \"class\": image_labels_vald}\nvalidation_data_frame = pd.DataFrame(validation_list)\nprint(validation_data_frame.head(5))\n# Write the validation labels to the disk.\nvalidation_data_frame.to_csv(os.path.join(SAVE_PATH, 'validation_list.csv'), columns=[\"Image_Name\", \"class\"], index=False)\n\n# Get the class names from the training folder.\nimage_class_names = os.listdir(IMAGE_PATH_TRAIN)\nimage_class_names.remove('.DS_Store') if '.DS_Store' in image_class_names else None\n\n# Create a new directory for organized validation images.\nvalidation_folder_path = os.path.join(SAVE_PATH, 'Validation_folder')\nos.makedirs(validation_folder_path, exist_ok=True)\n\nfor class_name in image_class_names:\n    os.makedirs(os.path.join(validation_folder_path, class_name), exist_ok=True)\n\n# Move the validation dataset to their subfolders according to their classes.\n# for idx, row in validation_data_frame.iterrows():\n#     move(os.path.join(IMAGE_PATH_VALID, row['Image_Name']), os.path.join(validation_folder_path, row['class'], row['Image_Name']))\n    \n","metadata":{"execution":{"iopub.status.busy":"2024-06-08T11:47:02.793365Z","iopub.execute_input":"2024-06-08T11:47:02.793816Z","iopub.status.idle":"2024-06-08T11:51:33.047504Z","shell.execute_reply.started":"2024-06-08T11:47:02.793767Z","shell.execute_reply":"2024-06-08T11:51:33.046684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read CSV\ntrain_labels = pd.read_csv('/kaggle/input/imagenet-object-localization-challenge/ILSVRC/ImageSets/CLS-LOC/train_cls.txt', delim_whitespace=True, header=None, names=['Image_Name', 'class'])\nvalid_labels = pd.read_csv('/kaggle/working/validation_list.csv')","metadata":{"execution":{"iopub.status.busy":"2024-06-08T11:51:33.048824Z","iopub.execute_input":"2024-06-08T11:51:33.049246Z","iopub.status.idle":"2024-06-08T11:51:34.520854Z","shell.execute_reply.started":"2024-06-08T11:51:33.049215Z","shell.execute_reply":"2024-06-08T11:51:34.519999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_labels.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-08T11:51:34.523417Z","iopub.execute_input":"2024-06-08T11:51:34.52389Z","iopub.status.idle":"2024-06-08T11:51:34.536837Z","shell.execute_reply.started":"2024-06-08T11:51:34.523856Z","shell.execute_reply":"2024-06-08T11:51:34.535926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import tensorflow as tf \n# from tensorflow.keras.applications import ResNet50  \n# import numpy as np\n# # import tensorflow_datasets as tfds\n# import seaborn as sns\n# import matplotlib.pyplot as plt\n# from tqdm import tqdm\n# import os\n# import random\n# from glob import glob\n\n# from keras.models import *\n# from keras.layers import *\n# from tensorflow.keras.callbacks import EarlyStopping\n# from tensorflow.keras.optimizers import Adam\n\n# from PIL import Image","metadata":{"execution":{"iopub.status.busy":"2024-06-08T10:02:52.78032Z","iopub.execute_input":"2024-06-08T10:02:52.780761Z","iopub.status.idle":"2024-06-08T10:02:52.788311Z","shell.execute_reply.started":"2024-06-08T10:02:52.780731Z","shell.execute_reply":"2024-06-08T10:02:52.787434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tensorflow usefull import\nfrom keras.applications.vgg16 import preprocess_input\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator,load_img, img_to_array,array_to_img\nfrom IPython.display import display\n\nmapping_path = '/kaggle/input/imagenet-object-localization-challenge/LOC_synset_mapping.txt'\n\n# Creating of mapping dictionaries to get the image classes\n\nclass_mapping_dict = {}\nclass_mapping_dict_number = {}\nmapping_class_to_number = {}\nmapping_number_to_class = {}\ni = 0\nfor line in open(mapping_path):\n    class_mapping_dict[line[:9].strip()] = line[9:].strip()\n    class_mapping_dict_number[i] = line[9:].strip()\n    mapping_class_to_number[line[:9].strip()] = i\n    mapping_number_to_class[i] = line[:9].strip()\n    i+=1\ntrain_path = '/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train'\n# valid_path = '/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/val'\n\n# Creation of dataset_array and true_classes\n\ndataset_array = []\ntrue_classes = []\nimages_array = []\nfor train_class in tqdm(os.listdir(train_path)):\n    i = 0\n    for el in os.listdir(train_path + '/' + train_class):\n        if i < 10:\n            path = train_path + '/' + train_class + '/' + el\n            image = load_img(path,target_size=(224,224,3))\n            image_array = img_to_array(image).astype(np.uint8)\n            images_array.append(image_array)\n            true_class = class_mapping_dict[path.split('/')[-2]]\n            true_classes.append(true_class)\n            i+=1\n        else:\n            break\nimages_array = np.array(images_array)\ntrue_classes = np.array(true_classes)\nprint('Preprocessing in progress')\ndataset_array = preprocess_input(images_array)\nprint('FINISH')\n\n# Print Imagenet samples befores and After Augmentation\nrand_indices = random.sample(range(0, 1000),5)\nplt.figure(figsize=(20, 20))\nplt.suptitle('Before preprocess',x = 0.5,y = 0.6)\nfor i in range(5):\n    ax = plt.subplot(1,5,i+1)\n    ax.imshow(images_array[rand_indices[i]])\nplt.figure(figsize=(20, 20))\nplt.suptitle('After preprocess',x = 0.5,y = 0.6)\nfor i in range(5):\n    ax = plt.subplot(1,5,i+1)\n    ax.imshow(dataset_array[rand_indices[i]])","metadata":{"execution":{"iopub.status.busy":"2024-06-08T10:41:42.119916Z","iopub.execute_input":"2024-06-08T10:41:42.120572Z","iopub.status.idle":"2024-06-08T10:42:25.620724Z","shell.execute_reply.started":"2024-06-08T10:41:42.120541Z","shell.execute_reply":"2024-06-08T10:42:25.619654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Same Implementation of loarding, Augmentation and Display of Data but only with pytorch**","metadata":{}},{"cell_type":"code","source":"# Define transformations\ntransform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\n\n# Create class mappings\nmapping_path = '/kaggle/input/imagenet-object-localization-challenge/LOC_synset_mapping.txt'\n\nclass_mapping_dict = {}\nclass_mapping_dict_number = {}\nmapping_class_to_number = {}\nmapping_number_to_class = {}\ni = 0\nfor line in open(mapping_path):\n    class_mapping_dict[line[:9].strip()] = line[9:].strip()\n    class_mapping_dict_number[i] = line[9:].strip()\n    mapping_class_to_number[line[:9].strip()] = i\n    mapping_number_to_class[i] = line[:9].strip()\n    i += 1\n\ntrain_path = '/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train'\n\n# Load and preprocess dataset\ndataset_array = []\ntrue_classes = []\nimages_array = []\n\nfor train_class in tqdm(os.listdir(train_path)):\n    i = 0\n    for el in os.listdir(os.path.join(train_path, train_class)):\n        if i < 10:\n            path = os.path.join(train_path, train_class, el)\n            image = Image.open(path).convert('RGB')\n            image_resized = transform(image)\n            images_array.append(image_resized)\n            true_class = class_mapping_dict[train_class]\n            true_classes.append(true_class)\n            i += 1\n        else:\n            break\n\nimages_array = torch.stack(images_array)\ntrue_classes = np.array(true_classes)\nprint('Preprocessing in progress')\ndataset_array = images_array  # They are already preprocessed in the transform step\nprint('FINISH')\n\n# Randomly sample 5 images\nrand_indices = random.sample(range(len(images_array)), 5)\n\n# Visualize images after preprocessing\nplt.figure(figsize=(20, 20))\nplt.suptitle('Augmented data', x=0.5, y=0.6)\nfor i, idx in enumerate(rand_indices):\n    ax = plt.subplot(1, 5, i + 1)\n    ax.imshow(dataset_array[idx].permute(1, 2, 0).numpy())\n    ax.axis('off')\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-08T14:12:52.49723Z","iopub.execute_input":"2024-06-08T14:12:52.498166Z","iopub.status.idle":"2024-06-08T14:22:14.24615Z","shell.execute_reply.started":"2024-06-08T14:12:52.49813Z","shell.execute_reply":"2024-06-08T14:22:14.245123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Define the model","metadata":{}},{"cell_type":"code","source":"from torch.utils.data import Dataset, DataLoader\n\nclass ImageNetDataset(Dataset):\n    def __init__(self, root_dir, class_mapping_dict, transform=None):\n        self.root_dir = root_dir\n        self.transform = transform\n        self.class_mapping_dict = class_mapping_dict\n        self.images = []\n        self.labels = []\n        \n        for train_class in tqdm(os.listdir(root_dir)):\n            class_path = os.path.join(root_dir, train_class)\n            for img_name in os.listdir(class_path):\n                self.images.append(os.path.join(class_path, img_name))\n                self.labels.append(self.class_mapping_dict[train_class])\n    \n    def __len__(self):\n        return len(self.images)\n    \n    def __getitem__(self, idx):\n        img_path = self.images[idx]\n        image = Image.open(img_path).convert('RGB')\n        label = self.labels[idx]\n        \n        if self.transform:\n            image = self.transform(image)\n        \n        return image, label\n","metadata":{"execution":{"iopub.status.busy":"2024-06-08T14:23:54.614234Z","iopub.execute_input":"2024-06-08T14:23:54.615135Z","iopub.status.idle":"2024-06-08T14:23:54.624192Z","shell.execute_reply.started":"2024-06-08T14:23:54.6151Z","shell.execute_reply":"2024-06-08T14:23:54.623229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the class mappings\nmapping_path = '/kaggle/input/imagenet-object-localization-challenge/LOC_synset_mapping.txt'\n\nclass_mapping_dict = {}\nfor line in open(mapping_path):\n    class_mapping_dict[line[:9].strip()] = line[9:].strip()\n\n# Define dataset paths\ntrain_path = '/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train'\n# val_path = '/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/val'\n\n# Create dataset instances\ntrain_dataset = ImageNetDataset(train_path, class_mapping_dict, transform)\n# val_dataset = ImageNetDataset(val_path, class_mapping_dict, transform)\n\n# Create DataLoader instances\ntrain_loader = DataLoader(train_dataset, batch_size=32, shuffle=True, num_workers=4)\n# val_loader = DataLoader(val_dataset, batch_size=32, shuffle=False, num_workers=4)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-08T14:25:43.543102Z","iopub.execute_input":"2024-06-08T14:25:43.543744Z","iopub.status.idle":"2024-06-08T14:25:47.420619Z","shell.execute_reply.started":"2024-06-08T14:25:43.543713Z","shell.execute_reply":"2024-06-08T14:25:47.418673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class AlexNet(nn.Module):\n\n    def __init__(self, num_classes=1000):\n        super(AlexNet, self).__init__()\n        self.layer1 = nn.Sequential(\n            nn.Conv2d(3, 96, kernel_size=11, stride=4, padding=0),\n            nn.ReLU(),\n            nn.LocalResponseNorm(size=5, alpha=0.0001, beta=0.75, k=2),\n            nn.MaxPool2d(kernel_size = 3, stride = 2))\n        \n        self.layer2 = nn.Sequential(\n            nn.Conv2d(96, 256, kernel_size=5, stride=1, padding=2),\n            nn.ReLU(),\n            nn.LocalResponseNorm(size=5, alpha=0.0001, beta=0.75, k=2),\n            nn.MaxPool2d(kernel_size = 3, stride = 2))\n        \n        self.layer3 = nn.Sequential(\n            nn.Conv2d(in_channels=256, out_channels=384, kernel_size=3, padding=1),\n            nn.ReLU())\n        \n        self.layer4 = nn.Sequential(\n            nn.Conv2d(in_channels=384, out_channels=384, kernel_size=3, padding=1),\n            nn.ReLU())\n        \n        self.layer5 = nn.Sequential(\n            nn.Conv2d(in_channels=384, out_channels=256, kernel_size=3, padding=1),\n            nn.ReLU(),\n            nn.MaxPool2d(kernel_size=3, stride=2))\n        \n        self.fc1 = nn.Sequential(\n            nn.Dropout(0.5),\n            nn.Linear(256 * 6 * 6, 4096),\n            nn.ReLU())\n        \n        self.fc2 = nn.Sequential(\n            nn.Dropout(0.5),\n            nn.Linear(4096, 4096),\n            nn.ReLU())\n        \n        self.fc3= nn.Sequential(\n            nn.Linear(4096, num_classes))\n        \n    def forward(self, x):\n\n        x = self.layer1(x)\n        x = self.layer2(x)\n        x = self.layer3(x)\n        x = self.layer4(x)\n        x = self.layer5(x)\n        \n        x = x.view(x.size(0), -1)\n\n        x = self.fc1(x)\n        x = self.fc2(x)\n        x = self.fc3(x)\n        \n        return x\n        \n","metadata":{"execution":{"iopub.status.busy":"2024-06-08T14:26:09.615223Z","iopub.execute_input":"2024-06-08T14:26:09.615581Z","iopub.status.idle":"2024-06-08T14:26:09.663342Z","shell.execute_reply.started":"2024-06-08T14:26:09.615554Z","shell.execute_reply":"2024-06-08T14:26:09.662475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define loss and optimizer\ncriterion = nn.CrossEntropyLoss()\n# optimizer = optims.SGD(model.parameters(), lr=0.01, momentum=0.9, weight_decay=0.0005)\nmodel = AlexNet(num_classes=len(class_mapping_dict))","metadata":{"execution":{"iopub.status.busy":"2024-06-08T14:26:15.855051Z","iopub.execute_input":"2024-06-08T14:26:15.855434Z","iopub.status.idle":"2024-06-08T14:26:16.330623Z","shell.execute_reply.started":"2024-06-08T14:26:15.855402Z","shell.execute_reply":"2024-06-08T14:26:16.329767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Train the model","metadata":{}},{"cell_type":"code","source":"def train(model, criterion, train_loader, num_epochs, initial_lr=0.01, adjust_lr_factor=10):\n    \"\"\"Training loop for a model with manual learning rate adjustment.\"\"\"\n    \n    # Define the optimizer with the initial learning rate\n    optimizer = optim.SGD(model.parameters(), lr=initial_lr, momentum=0.9, weight_decay=0.0005)\n    \n    # Track the best validation loss to adjust learning rate\n    best_val_loss = float('inf')\n    lr = initial_lr\n    print(\"Hell\")\n    for epoch in range(num_epochs):\n        model.train()  # Set model to training mode\n        running_loss = 0.0\n        print(\"Hello\", epoch+1)\n        \n        for images, labels in train_loader:\n            print(\"target:\", labels)\n            images, labels = images.to(device), labels.to(device)\n            print(\"Shape:\", images.size())\n            optimizer.zero_grad()  # Zero the parameter gradients\n            outputs = model(images)\n            loss = criterion(outputs, labels)\n            loss.backward()\n            optimizer.step()\n            \n            running_loss += loss.item()\n        \n        # Calculate average training loss\n        avg_train_loss = running_loss / len(train_loader)\n        print(f\"Epoch {epoch+1}/{num_epochs},Train Loss: {avg_train_loss:.4f}\")\n        # Evaluate on the validation set\n#         val_loss = validate(model, criterion, val_loader)\n        \n        # Adjust learning rate if validation loss does not improve\n#         if val_loss < best_val_loss:\n#             best_val_loss = val_loss\n#         else:\n#             lr /= adjust_lr_factor\n#             for param_group in optimizer.param_groups:\n#                 param_group['lr'] = lr\n\n          \n\n#         print(f\"Epoch {epoch+1}/{num_epochs},\n#               Train Loss: {avg_train_loss:.4f}, \n# #               Val Loss: {val_loss:.4f}, LR: {lr}\n#               \")\n\n# def validate(model, criterion, val_loader):\n#     \"\"\"Evaluate the model on the validation set.\"\"\"\n#     model.eval()  # Set model to evaluation mode\n#     val_loss = 0.0\n    \n#     test_size = len(val_loader.dataset)\n#     correct = 0\n    \n#     with torch.no_grad():\n#         for images, labels in val_loader:\n#             images, labels = images.to(device), labels.to(device)\n#             outputs = model(images)\n#             loss = criterion(outputs, labels)\n#             val_loss += loss.item()\n            \n#             _, preds = torch.max(output, 1)\n#             correct += (preds == labels).sum().item()\n    \n#     avg_val_loss = val_loss / len(val_loader)\n    \n#     accuracy = correct / test_size * 100\n#     print(f'Test Accuracy: {accuracy:.2f}%')\n#     return avg_val_loss\n\n# # Assume you have your DataLoader objects: train_loader and val_loader\nnum_epochs = 10  # Number of epochs/\ntrain(model, criterion, train_loader, num_epochs)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-08T15:03:41.615006Z","iopub.execute_input":"2024-06-08T15:03:41.61598Z","iopub.status.idle":"2024-06-08T15:03:43.688078Z","shell.execute_reply.started":"2024-06-08T15:03:41.615941Z","shell.execute_reply":"2024-06-08T15:03:43.686621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport torch\nimport torchvision as tv\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nimport torchvision.transforms as transforms\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nfrom glob import glob\nfrom torch.utils.data import Dataset, DataLoader\nimport os\nimport xml.etree.ElementTree as ET\n\n# Device configuration\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\n# Path constants for Kaggle environment\nIMAGE_PATH_VALID = '/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/val/'\nANNOTATION_PATH = '/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Annotations/CLS-LOC/val/'\nIMAGE_PATH_TRAIN = '/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train/'\nSAVE_PATH = '/kaggle/working/'\n\n# Get the class names from the training folder.\nimage_class_names = os.listdir(IMAGE_PATH_TRAIN)\nif '.DS_Store' in image_class_names:\n    image_class_names.remove('.DS_Store')\n\n# Define transformations\ntransform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\n\n# Load class mappings\nmapping_path = '/kaggle/input/imagenet-object-localization-challenge/LOC_synset_mapping.txt'\nclass_mapping_dict = {}\nmapping_class_to_number = {}\nmapping_number_to_class = {}\ni = 0\nfor line in open(mapping_path):\n    class_mapping_dict[line[:9].strip()] = line[9:].strip()\n    mapping_class_to_number[line[:9].strip()] = i\n    mapping_number_to_class[i] = line[:9].strip()\n    i += 1\n\nclass ImageNetDataset(Dataset):\n    def __init__(self, root_dir, class_mapping_dict, transform=None):\n        self.root_dir = root_dir\n        self.transform = transform\n        self.class_mapping_dict = class_mapping_dict\n        self.images = []\n        self.labels = []\n        \n        for train_class in tqdm(os.listdir(root_dir)):\n            class_path = os.path.join(root_dir, train_class)\n            for img_name in os.listdir(class_path):\n                self.images.append(os.path.join(class_path, img_name))\n                self.labels.append(mapping_class_to_number[train_class])\n    \n    def __len__(self):\n        return len(self.images)\n    \n    def __getitem__(self, idx):\n        img_path = self.images[idx]\n        image = Image.open(img_path).convert('RGB')\n        label = self.labels[idx]\n        \n        if self.transform:\n            image = self.transform(image)\n        \n        return image, label\n\n# Create dataset instances\ntrain_dataset = ImageNetDataset(IMAGE_PATH_TRAIN, class_mapping_dict, transform)\n# val_dataset = ImageNetDataset(IMAGE_PATH_VALID, class_mapping_dict, transform)\n\n# Create DataLoader instances\ntrain_loader = DataLoader(train_dataset, batch_size=32, shuffle=True, num_workers=4)\n# val_loader = DataLoader(val_dataset, batch_size=32, shuffle=False, num_workers=4)\n\n# class AlexNet(nn.Module):\n#     def __init__(self, num_classes=1000):\n#         super(AlexNet, self).__init__()\n#         self.layer1 = nn.Sequential(\n#             nn.Conv2d(3, 96, kernel_size=11, stride=4, padding=0),\n#             nn.ReLU(),\n#             nn.LocalResponseNorm(size=5, alpha=0.0001, beta=0.75, k=2),\n#             nn.MaxPool2d(kernel_size=3, stride=2))\n        \n#         self.layer2 = nn.Sequential(\n#             nn.Conv2d(96, 256, kernel_size=5, stride=1, padding=2),\n#             nn.ReLU(),\n#             nn.LocalResponseNorm(size=5, alpha=0.0001, beta=0.75, k=2),\n#             nn.MaxPool2d(kernel_size=3, stride=2))\n        \n#         self.layer3 = nn.Sequential(\n#             nn.Conv2d(256, 384, kernel_size=3, padding=1),\n#             nn.ReLU())\n        \n#         self.layer4 = nn.Sequential(\n#             nn.Conv2d(384, 384, kernel_size=3, padding=1),\n#             nn.ReLU())\n        \n#         self.layer5 = nn.Sequential(\n#             nn.Conv2d(384, 256, kernel_size=3, padding=1),\n#             nn.ReLU(),\n#             nn.MaxPool2d(kernel_size=3, stride=2))\n        \n#         self.fc1 = nn.Sequential(\n#             nn.Dropout(0.5),\n#             nn.Linear(256 * 6 * 6, 4096),\n#             nn.ReLU())\n        \n#         self.fc2 = nn.Sequential(\n#             nn.Dropout(0.5),\n#             nn.Linear(4096, 4096),\n#             nn.ReLU())\n        \n#         self.fc3 = nn.Sequential(\n#             nn.Linear(4096, num_classes))\n        \n#     def forward(self, x):\n#         x = self.layer1(x)\n#         x = self.layer2(x)\n#         x = self.layer3(x)\n#         x = self.layer4(x)\n#         x = self.layer5(x)\n#         x = x.view(x.size(0), -1)\n#         x = self.fc1(x)\n#         x = self.fc2(x)\n#         x = self.fc3(x)\n#         return x\n\n# Define loss and optimizer\nmodel = AlexNet(num_classes=len(class_mapping_dict)).to(device)\ncriterion = nn.CrossEntropyLoss()\n\ndef train(model, criterion, train_loader, num_epochs, initial_lr=0.01, adjust_lr_factor=10):\n    \"\"\"Training loop for a model with manual learning rate adjustment.\"\"\"\n    \n    optimizer = optim.SGD(model.parameters(), lr=initial_lr, momentum=0.9, weight_decay=0.0005)\n    best_val_loss = float('inf')\n    lr = initial_lr\n    \n    for epoch in range(num_epochs):\n        model.train()  # Set model to training mode\n        running_loss = 0.0\n        \n        for batch_idx, (images, labels) in enumerate(train_loader):\n            images, labels = images.to(device), labels.to(device)\n            optimizer.zero_grad()  # Zero the parameter gradients\n            outputs = model(images)\n            loss = criterion(outputs, labels)\n            loss.backward()\n            optimizer.step()\n            running_loss += loss.item()\n            print(batch_idx)\n        \n        avg_train_loss = running_loss / len(train_loader)\n        print(f\"Epoch {epoch+1}/{num_epochs}, Train Loss: {avg_train_loss:.4f}\")\n        \n#         val_loss = validate(model, criterion, val_loader)\n        \n#         if val_loss < best_val_loss:\n#             best_val_loss = val_loss\n#         else:\n#             lr /= adjust_lr_factor\n#             for param_group in optimizer.param_groups:\n#                 param_group['lr'] = lr\n#         print(f\"Epoch {epoch+1}/{num_epochs}, Val Loss: {val_loss:.4f}, LR: {lr:.6f}\")\n\ndef validate(model, criterion, val_loader):\n    \"\"\"Evaluate the model on the validation set.\"\"\"\n    model.eval()  # Set model to evaluation mode\n    val_loss = 0.0\n    correct = 0\n    total = 0\n    \n    with torch.no_grad():\n        for images, labels in val_loader:\n            images, labels = images.to(device), labels.to(device)\n            outputs = model(images)\n            loss = criterion(outputs, labels)\n            val_loss += loss.item()\n            _, predicted = torch.max(outputs, 1)\n            total += labels.size(0)\n            correct += (predicted == labels).sum().item()\n    \n    avg_val_loss = val_loss / len(val_loader)\n    accuracy = correct / total * 100\n    print(f'Validation Accuracy: {accuracy:.2f}%')\n    return avg_val_loss\n\nnum_epochs = 10  # Number of epochs\ntrain(model, criterion, train_loader, num_epochs)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-08T18:30:01.135535Z","iopub.execute_input":"2024-06-08T18:30:01.135939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Get the top five of the model...","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Intermediate Code","metadata":{}},{"cell_type":"code","source":"def resize_image(src, size=(128, 128), bgc=\"white\"):\n    src.thumbnail(size, Image.ANTIALIAS)\n    \n    new_image = Image.new(\"RGB\", size, bgc)\n    \n    new_image.paste(src, (int((size[0]-src.size[0]) / 2)), int((size[1] - src.size[1]) / 2))\n    \n    return new_image","metadata":{"execution":{"iopub.status.busy":"2024-06-07T22:02:04.22196Z","iopub.execute_input":"2024-06-07T22:02:04.223131Z","iopub.status.idle":"2024-06-07T22:02:04.229843Z","shell.execute_reply.started":"2024-06-07T22:02:04.22309Z","shell.execute_reply":"2024-06-07T22:02:04.228655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transform = transforms.Compose([\n    transforms.Resize([256, 256]),\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomVerticalFlip(),\n    transforms.ColorJitter(brightness=0.5, contrast=0),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.5, 0.5, 0.5], std=[0.5, 0.5, 0.5])\n])","metadata":{"execution":{"iopub.status.busy":"2024-06-07T22:02:05.505941Z","iopub.execute_input":"2024-06-07T22:02:05.50634Z","iopub.status.idle":"2024-06-07T22:02:05.513367Z","shell.execute_reply.started":"2024-06-07T22:02:05.506306Z","shell.execute_reply":"2024-06-07T22:02:05.512067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load(path):\n    dataset = tv.datasets.ImageFolder(root=path,transform = transform)\n    \n    train_size = int(0.7 * len(dataset))\n    test_size = len(dataset) - train_size\n    \n    train_set, test_set = torch.utils.data.random_split(dataset, [train_size, test_size])\n    \n    train_loader = torch.utils.data.DataLoader(\n        train_set,\n        batch_size=50,\n        num_workers=0,\n        shuffle=False\n    )\n        valid_loader = torch.utils.data.DataLoader(\n        train_set,\n        batch_size=50,\n        num_workers=0,\n        shuffle=False\n    )\n    \n    test_loader = torch.utils.data.DataLoader(\n        test_set,\n        batch_size=50,\n        num_workers=0,\n        shuffle=False\n    )\n\n    return train_loader, test_loader\n\ntest_loader, train_loader, valid_loader = load('/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC')\nprint(f\"train loaders: {train_loader}\")","metadata":{"execution":{"iopub.status.busy":"2024-06-07T22:02:40.561111Z","iopub.execute_input":"2024-06-07T22:02:40.562038Z","iopub.status.idle":"2024-06-07T22:03:01.03644Z","shell.execute_reply.started":"2024-06-07T22:02:40.561994Z","shell.execute_reply":"2024-06-07T22:03:01.034806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(device)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nmodel = tv.models.resnet50(weights=tv.models.ResNet50_Weights.DEFAULT)\n\nmodel = model.to(device)\n\ncriterion = torch.nn.CrossEntropyLoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=0.05)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss = []\n\nfor epoch, (inputs, labels) in enumerate(train_loader):\n    print(f'Epoch {epoch+1}')  \n          \n    # Move input and label tensors to the device\n    inputs = inputs.to(device)\n    labels = labels.to(device)\n\n    # Zero out the optimizer\n    optimizer.zero_grad()\n\n    # Forward pass\n    outputs = model(inputs)\n    print(f'Outputs: {outputs}')\n    loss.append(criterion(outputs, labels))\n\n    # Backward pass\n    loss[epoch].backward()\n    optimizer.step()\n\n    # Print the loss for every epoch\n    print(f'Loss: {loss[epoch].item():.4f}')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torchvision.transforms as transforms\nimport torchvision.datasets as datasets\nfrom PIL import Image\nimport torch.nn.init as init\nimport torch.optim as optim\nimport os","metadata":{"id":"DiotJrQTjgua"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the data augmentation transformations\ndata_transforms = transforms.Compose([\n    # Randomly resize and crop the image to 224x224\n    transforms.RandomResizedCrop(224),\n    # Randomly flip the image horizontally\n    transforms.RandomHorizontalFlip(),\n    # Randomly adjust brightness, contrast, saturation, and hue\n    transforms.ColorJitter(brightness=0.2, contrast=0.2, saturation=0.2, hue=0.1),\n    # Convert the image to a PyTorch tensor\n    transforms.ToTensor(),\n])\n\n# Define paths to save the model\nsave_dir = \"./models\"\nos.makedirs(save_dir, exist_ok=True)\n\n# Define data loaders for training and validation sets\ntrain_dataset = datasets.ImageNet(root=\"path/to/ImageNet/train\", split='train', transform=data_transforms)\nval_dataset = datasets.ImageNet(root=\"path/to/ImageNet/val\", split='val', transform=data_transforms)\n\ntrain_loader = torch.utils.data.DataLoader(train_dataset, batch_size=64, shuffle=True, num_workers=4)\nval_loader = torch.utils.data.DataLoader(val_dataset, batch_size=64, shuffle=False, num_workers=4)\n","metadata":{"id":"c5xNR0iLjUCw"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class AlexNet(nn.Module):\n    def __init__(self, num_classes=1000):\n        super(AlexNet, self).__init__()\n        self.features = nn.Sequential(\n            nn.Conv2d(3, 64, kernel_size=11, stride=4, padding=2),\n            nn.ReLU(inplace=True),\n            nn.LocalResponseNorm(size=5, alpha=0.0001, beta=0.75, k=2),  # LRN after conv1\n            nn.MaxPool2d(kernel_size=3, stride=2),\n\n            nn.Conv2d(64, 192, kernel_size=5, padding=2),\n            nn.ReLU(inplace=True),\n            nn.LocalResponseNorm(size=5, alpha=0.0001, beta=0.75, k=2),  # LRN after conv2\n            nn.MaxPool2d(kernel_size=3, stride=2),\n\n            nn.Conv2d(192, 384, kernel_size=3, padding=1),\n            nn.ReLU(inplace=True),\n\n            nn.Conv2d(384, 256, kernel_size=3, padding=1),\n            nn.ReLU(inplace=True),\n\n            nn.Conv2d(256, 256, kernel_size=3, padding=1),\n            nn.ReLU(inplace=True),\n            nn.MaxPool2d(kernel_size=3, stride=2),\n        )\n\n        self.avgpool = nn.AdaptiveAvgPool2d((6, 6))\n\n        self.classifier = nn.Sequential(\n            nn.Dropout(),\n            nn.Linear(256 * 6 * 6, 4096),\n            nn.ReLU(inplace=True),\n            nn.Dropout(),\n            nn.Linear(4096, 4096),\n            nn.ReLU(inplace=True),\n            nn.Linear(4096, num_classes),\n        )\n\n    def forward(self, x):\n        x = self.features(x)\n        x = torch.flatten(x, 1)\n        x = self.classifier(x)\n        return x\n\n    def _initialize_weights(self):\n        for m in self.modules():\n            if isinstance(m, nn.Conv2d) or isinstance(m, nn.Linear):\n                nn.init.normal_(m.weight, 0, 0.01)\n                if m.bias is not None:\n                    nn.init.constant_(m.bias, 1 if isinstance(m, nn.Conv2d) and m in [self.features[1], self.features[4], self.features[7]] else 0)\n\n","metadata":{"id":"35MdKDInjZCq","execution":{"iopub.status.busy":"2024-06-08T18:29:49.113965Z","iopub.execute_input":"2024-06-08T18:29:49.114367Z","iopub.status.idle":"2024-06-08T18:29:49.128083Z","shell.execute_reply.started":"2024-06-08T18:29:49.114335Z","shell.execute_reply":"2024-06-08T18:29:49.1271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Instantiate the model and initialize weights\nmodel = AlexNet(num_classes=1000)\nmodel._initialize_weights()\n\n# Define loss function and optimizer\ncriterion = nn.CrossEntropyLoss()\noptimizer = optim.SGD(model.parameters(), lr=0.01, momentum=0.9)\n\n# Move model to GPU if available\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel.to(device)\n\n# Training loop\nnum_epochs = 5\ntotal_step = len(train_loader)\n\nfor epoch in range(num_epochs):\n    model.train()\n    running_loss = 0.0\n    correct_train = 0\n    total_train = 0\n\n    for i, (images, labels) in enumerate(train_loader):\n        # Move tensors to the configured device\n        images = images.to(device)\n        labels = labels.to(device)\n\n        # Forward pass\n        outputs = model(images)\n        loss = criterion(outputs, labels)\n\n        # Backward and optimize\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n\n        running_loss += loss.item()\n        _, predicted = torch.max(outputs.data, 1)\n        total_train += labels.size(0)\n        correct_train += (predicted == labels).sum().item()\n\n    train_loss = running_loss / total_step\n    train_acc = correct_train / total_train\n\n    # Validation\n    model.eval()\n    val_loss = 0.0\n    correct_val = 0\n    total_val = 0\n    with torch.no_grad():\n        for images, labels in val_loader:\n            images = images.to(device)\n            labels = labels.to(device)\n            outputs = model(images)\n            loss = criterion(outputs, labels)\n            val_loss += loss.item()\n            _, predicted = torch.max(outputs.data, 1)\n            total_val += labels.size(0)\n            correct_val += (predicted == labels).sum().item()\n\n    val_loss = val_loss / len(val_loader)\n    val_acc = correct_val / total_val\n\n    print('Epoch [{}/{}], Train Loss: {:.4f}, Train Acc: {:.4f}, Val Loss: {:.4f}, Val Acc: {:.4f}'\n          .format(epoch+1, num_epochs, train_loss, train_acc, val_loss, val_acc))\n\n    # Save the model after each epoch\n    save_path = os.path.join(save_dir, f\"alexnet_epoch_{epoch + 1}.pt\")\n    torch.save(model.state_dict(), save_path)\n    print(f\"Model saved at: {save_path}\")","metadata":{"id":"PSDZgvHxjblX","execution":{"iopub.status.busy":"2024-06-08T14:57:31.587187Z","iopub.execute_input":"2024-06-08T14:57:31.587957Z","iopub.status.idle":"2024-06-08T14:57:32.608094Z","shell.execute_reply.started":"2024-06-08T14:57:31.587926Z","shell.execute_reply":"2024-06-08T14:57:32.606894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}