{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import torch\nimport numpy as np\nfrom torch.utils.data import Dataset, TensorDataset, SubsetRandomSampler\nimport glob\nimport cv2 as cv\nimport torchvision.transforms as T\nimport random\nimport pandas as pd\nimport torch.nn as nn\nfrom tqdm import tqdm\nimport time\nimport shutil\nimport json\nimport torchvision.models as models\n\ntorch.manual_seed(0)\nnp.random.seed(0)\n\n\ndef create_annotation_file(path, out_file=\"data.csv\"):\n    open(out_file, \"w\").close()\n    hotel_ids = glob.glob(pathname=path + \"/train_images/*\")\n    for id in hotel_ids:\n        y = id.split(\"/\")[-1]\n        imgs = glob.glob(pathname=id + \"/*\")\n        with open(out_file, \"a\") as f:\n            for img in imgs:\n                f.write(\",\".join([img.split(\"/\")[-1], y]) + \"\\n\")\n            f.close()\n\n\nclass HotelDataset(Dataset):\n    def __init__(self, path, annotation_file_path, train=True, transform=None):\n        self.transform = transform\n        self.train = train\n        self.img_path = path\n        if self.train:\n            self.annotation = pd.read_csv(annotation_file_path)#.iloc[:, :]\n            self.unique_hotel_ids = self.annotation.iloc[:, 1].unique().tolist()\n            self.translate = {idx: x for idx,x in enumerate(self.unique_hotel_ids)}\n            with open(\"label_to_hotel_id.json\", \"w\") as f:\n                f.write(json.dumps(self.translate))\n                f.close()\n        else:\n            self.annotation = glob.glob(pathname=path + \"/test_images/*\")\n        \n        \n\n    def __getitem__(self, idx):\n        if self.train:\n            x = self.annotation.iloc[idx, 0]\n            y = self.annotation.iloc[idx, 1]\n            x = torch.tensor(cv.imread(\n                \"{}/{}/{}/{}\".format(self.img_path, \"train_images\", y, x))).permute(2, 0, 1)\n            label = self.unique_hotel_ids.index(int(y))\n            y = torch.tensor(label)\n            if self.transform is not None:\n                x = self.transform(x) / 255.\n            else:\n                x = x / 255.\n            return x, y\n        else:\n            x = self.annotation[idx]\n            x = torch.tensor(cv.imread(\n                \"{}/{}/{}\".format(self.img_path, \"test_images\", x))).permute(2, 0, 1)\n            x = x / 255.\n            return x\n\n    def __len__(self):\n        return len(self.annotation)  # \"Length X: {}, Length Y: {}\".format(len(self.x), len(self.y))\n\n\nif __name__ == '__main__':\n    create_annotation_file(\"../input/hotel-id-to-combat-human-trafficking-2022-fgvc9/\")\n    transform = T.Compose(\n        [\n            T.ConvertImageDtype(torch.uint8),\n            T.RandomGrayscale(p=0.2),\n            T.RandomHorizontalFlip(p=0.2),\n            T.RandomRotation(random.randint(0, 30)),\n            T.Resize(size=(256, 256)),\n\n        ]\n    )\n    train_dataset = HotelDataset(\"../input/hotel-id-to-combat-human-trafficking-2022-fgvc9/\",\n                           transform=transform, annotation_file_path=\"data.csv\")\n   \n    \n    validation_split = .2\n    # Creating data indices for training and validation splits:\n    dataset_size = len(train_dataset)\n    indices = list(range(dataset_size))\n    split = int(np.floor(validation_split * dataset_size))\n    np.random.shuffle(indices)\n    train_indices, val_indices = indices[split:], indices[:split]\n    train_sampler = SubsetRandomSampler(train_indices)\n    val_sampler = SubsetRandomSampler(val_indices)\n    train_loader = torch.utils.data.DataLoader(train_dataset, sampler=train_sampler, batch_size=32, num_workers=2)\n    val_loader = torch.utils.data.DataLoader(train_dataset, sampler=val_sampler, batch_size=32, num_workers=2)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-28T15:43:24.890216Z","iopub.execute_input":"2022-05-28T15:43:24.890921Z","iopub.status.idle":"2022-05-28T15:43:59.810989Z","shell.execute_reply.started":"2022-05-28T15:43:24.890808Z","shell.execute_reply":"2022-05-28T15:43:59.810251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Encoder(nn.Module):\n    def __init__(self):\n        super(Encoder, self).__init__()\n        self.model = nn.Sequential(\n            nn.Conv2d(3, 16, (5, 5)),\n            nn.BatchNorm2d(16),\n            nn.ReLU(),\n            nn.Conv2d(16, 32, (5, 5)),\n            nn.BatchNorm2d(32),\n            nn.ReLU(),\n            nn.MaxPool2d(3, 3),\n\n            nn.Conv2d(32, 64, (5, 5)),\n            nn.BatchNorm2d(64),\n            nn.ReLU(),\n            nn.Dropout2d(p=0.25),\n            nn.Conv2d(64, 128, (3, 3)),\n            nn.BatchNorm2d(128),\n            nn.ReLU(),\n            nn.MaxPool2d(3, 3),\n\n            nn.Conv2d(128, 256, (5, 5)),\n            nn.BatchNorm2d(256),\n            nn.ReLU(),\n            nn.MaxPool2d(3, 3),\n            \n#             nn.Conv2d(512, 512, (3, 3)),\n#             nn.BatchNorm2d(512),\n#             nn.ReLU(),\n#             nn.MaxPool2d(2, 2),\n            \n            nn.Flatten(),\n            nn.Linear(6*6*256, 3116),\n\n        )\n        \n        \n    def forward(self, x):\n        x = self.model(x)\n        print(x.shape)\n        return x\n\nresnet = models.resnet18(pretrained=True)\nresnet.fc = nn.Linear(resnet.fc.in_features, 3116)\nfor param in resnet.parameters():\n    param.requires_grad = False\nfor param in resnet.fc.parameters():\n    param.requires_grad = True","metadata":{"execution":{"iopub.status.busy":"2022-05-28T15:43:59.812864Z","iopub.execute_input":"2022-05-28T15:43:59.813356Z","iopub.status.idle":"2022-05-28T15:44:02.554237Z","shell.execute_reply.started":"2022-05-28T15:43:59.813312Z","shell.execute_reply":"2022-05-28T15:44:02.553449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save_checkpoint(state, is_best, filename='checkpoint.pth.tar'):\n    torch.save(state, filename)\n    if is_best:\n        shutil.copyfile(filename, 'model_best.pth.tar')\n\ndef val_step(model):\n    global best_val_loss\n    vals = []\n    model.to(device)\n    model.eval()\n    with torch.no_grad():\n        for batch_ndx, (x, y) in enumerate(val_loader):\n            x = x.to(device)\n            y = y.to(device)\n            y_pred = encoder(x)\n            loss = loss_fn(y_pred, y)\n            is_best = loss.cpu().item() < best_val_loss\n            best_val_loss = min(loss.cpu().item(), best_val_loss)\n            save_checkpoint({\n                'epoch': epoch + 1,\n                'batch': batch_ndx + 1,\n                'state_dict': model.state_dict(),\n                'best_val_loss': loss.cpu().item()}, is_best)\n            del y_pred, x, y\n            vals.append(loss.cpu().item())\n    return sum(vals)/len(vals)","metadata":{"execution":{"iopub.status.busy":"2022-05-28T15:44:02.555757Z","iopub.execute_input":"2022-05-28T15:44:02.556059Z","iopub.status.idle":"2022-05-28T15:44:02.566675Z","shell.execute_reply.started":"2022-05-28T15:44:02.556008Z","shell.execute_reply":"2022-05-28T15:44:02.565705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_val_loss = 99999999\ndevice = torch.device(\"cuda\") if torch.cuda.is_available() else torch.device(\"cpu\")\nloss_fn = nn.CrossEntropyLoss()\nencoder = resnet\nencoder.to(device)\nencoder = torch.nn.DataParallel(encoder)\noptimizer = torch.optim.Adam(encoder.parameters())\n\nfor epoch in range(50):\n    for batch_ndx, (x, y) in enumerate(train_loader):\n        encoder.train()\n        x = x.to(device)\n        y = y.to(device)\n        optimizer.zero_grad()\n        y_pred = encoder(x)\n        loss = loss_fn(y_pred, y)\n        loss.backward()\n        optimizer.step()\n        \n        if batch_ndx % 10 == 0:\n            val_loss = val_step(encoder)\n            print(\"Epoch: {}, Batch: {}, Loss: {:.4f}, Val Loss: {:.4f}\".format(epoch, batch_ndx, loss.cpu().item(), val_loss))\n        else:\n            print(\"Epoch: {}, Batch: {}, Loss: {:.4f}, Val Loss: N/A\".format(epoch, batch_ndx, loss.cpu().item()))\n       \n    \n\n","metadata":{"execution":{"iopub.status.busy":"2022-05-28T15:44:02.568984Z","iopub.execute_input":"2022-05-28T15:44:02.569332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}