{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport json\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-26T03:56:14.651790Z","iopub.execute_input":"2022-07-26T03:56:14.652567Z","iopub.status.idle":"2022-07-26T03:56:15.947559Z","shell.execute_reply.started":"2022-07-26T03:56:14.652471Z","shell.execute_reply":"2022-07-26T03:56:15.946574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dir = '../input/herbarium-2022-fgvc9/train_images/'\ntest_dir = '../input/herbarium-2022-fgvc9/test_images/'\n\nwith open(\"../input/herbarium-2022-fgvc9/train_metadata.json\") as json_file:\n    train_meta = json.load(json_file)\nwith open(\"../input/herbarium-2022-fgvc9/test_metadata.json\") as json_file:\n    test_meta = json.load(json_file)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T03:56:15.950644Z","iopub.execute_input":"2022-07-26T03:56:15.951343Z","iopub.status.idle":"2022-07-26T03:56:29.113192Z","shell.execute_reply.started":"2022-07-26T03:56:15.951295Z","shell.execute_reply":"2022-07-26T03:56:29.112118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_ids = [image[\"image_id\"] for image in train_meta[\"images\"]]\nimage_dirs = [train_dir + image['file_name'] for image in train_meta[\"images\"]]\ncategory_ids = [annotation['category_id'] for annotation in train_meta['annotations']]\ngenus_ids = [annotation['genus_id'] for annotation in train_meta['annotations']]\n\ntest_ids = [image['image_id'] for image in test_meta]\ntest_dirs = [test_dir + image['file_name'] for image in test_meta]\n\ntrain_df = pd.DataFrame({\n    \"image_id\" : image_ids,\n    \"image_dir\" : image_dirs,\n    \"category\" : category_ids,\n    \"genus\" : genus_ids})\n\ntest_df = pd.DataFrame({\n    \"test_id\" : test_ids,\n    \"test_dir\" : test_dirs\n})\n\n\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T03:56:29.114532Z","iopub.execute_input":"2022-07-26T03:56:29.114833Z","iopub.status.idle":"2022-07-26T03:56:30.190655Z","shell.execute_reply.started":"2022-07-26T03:56:29.114805Z","shell.execute_reply":"2022-07-26T03:56:30.189566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"genus_map = {genus['genus_id'] : genus['genus'] for genus in train_meta['genera']}\ntrain_df['genus'] = train_df['genus'].map(genus_map)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-07-26T03:56:30.192722Z","iopub.execute_input":"2022-07-26T03:56:30.193134Z","iopub.status.idle":"2022-07-26T03:56:30.235830Z","shell.execute_reply.started":"2022-07-26T03:56:30.193099Z","shell.execute_reply":"2022-07-26T03:56:30.234868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Top 15 Genus ')\nprint(train_df['genus'].value_counts().head(15))\nprint()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T03:56:30.237213Z","iopub.execute_input":"2022-07-26T03:56:30.237749Z","iopub.status.idle":"2022-07-26T03:56:30.283484Z","shell.execute_reply.started":"2022-07-26T03:56:30.237707Z","shell.execute_reply":"2022-07-26T03:56:30.282090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = train_df['genus'].value_counts().head(15)\ndata = pd.DataFrame({'Genus' : data.index,\n                     'values' : data.values})\nplt.figure(figsize = (20, 10))\nsns.barplot(x='values', y = 'Genus', data = data , palette='summer_r')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T03:56:30.285084Z","iopub.execute_input":"2022-07-26T03:56:30.285732Z","iopub.status.idle":"2022-07-26T03:56:30.684684Z","shell.execute_reply.started":"2022-07-26T03:56:30.285690Z","shell.execute_reply":"2022-07-26T03:56:30.683573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_images(speices):\n    images = train_df.loc[train_df['genus'] == speices]['image_dir'][:6]\n    i = 1\n    fig = plt.figure(figsize = (18, 18))\n    plt.suptitle(speices, fontsize = '30')\n    for image in images:\n        img = cv2.imread(image)\n        ax = fig.add_subplot(2, 3, i)\n        ax.imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))\n        ax.set_axis_off()\n        i += 1\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T03:56:30.686133Z","iopub.execute_input":"2022-07-26T03:56:30.686560Z","iopub.status.idle":"2022-07-26T03:56:30.694485Z","shell.execute_reply.started":"2022-07-26T03:56:30.686520Z","shell.execute_reply":"2022-07-26T03:56:30.693357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in data['Genus'][:10]:\n    show_images(i)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:00:24.675319Z","iopub.execute_input":"2022-07-26T04:00:24.675854Z","iopub.status.idle":"2022-07-26T04:00:39.038145Z","shell.execute_reply.started":"2022-07-26T04:00:24.675807Z","shell.execute_reply":"2022-07-26T04:00:39.036936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modelling","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torchvision\nimport torchvision.transforms as transforms\nfrom torch.utils.data import Dataset, DataLoader\nfrom PIL import Image","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:00:50.532401Z","iopub.execute_input":"2022-07-26T04:00:50.532811Z","iopub.status.idle":"2022-07-26T04:00:52.715380Z","shell.execute_reply.started":"2022-07-26T04:00:50.532776Z","shell.execute_reply":"2022-07-26T04:00:52.714227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH = 64\nEPOCHS = 8\n\nLR = 0.01\nIM_SIZE = 224\n\nX_Train, Y_Train = train_df['image_dir'].values, train_df['category'].values\n\nTransform = transforms.Compose(\n    [transforms.ToTensor(),\n    transforms.Resize((IM_SIZE, IM_SIZE)),\n    transforms.Normalize((0.485, 0.456, 0.406), (0.229, 0.224, 0.225))])","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:00:52.717455Z","iopub.execute_input":"2022-07-26T04:00:52.718054Z","iopub.status.idle":"2022-07-26T04:00:52.724789Z","shell.execute_reply.started":"2022-07-26T04:00:52.718022Z","shell.execute_reply":"2022-07-26T04:00:52.723972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class GetData(Dataset):\n    def __init__(self, FNames, Labels, Transform):\n        self.fnames = FNames\n        self.transform = Transform\n        self.labels = Labels         \n        \n    def __len__(self):\n        return len(self.fnames)\n\n    def __getitem__(self, index):       \n        x = Image.open(self.fnames[index])\n    \n        if \"train\" in self.fnames[index]:             \n            return self.transform(x), self.labels[index]\n        elif \"test\" in self.fnames[index]:            \n            return self.transform(x), self.fnames[index]\n                \ntrainset = GetData(X_Train, Y_Train, Transform)\ntrainloader = DataLoader(trainset, batch_size=BATCH, shuffle=True)\n\nN_Classes = train_df['category'].nunique()\nnext(iter(trainloader))[0].shape\n\ndevice = 'cuda:0' if torch.cuda.is_available() else 'cpu'\nmodel = torchvision.models.densenet169(pretrained=True)            ","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:00:52.725912Z","iopub.execute_input":"2022-07-26T04:00:52.726514Z","iopub.status.idle":"2022-07-26T04:01:01.983833Z","shell.execute_reply.started":"2022-07-26T04:00:52.726479Z","shell.execute_reply":"2022-07-26T04:01:01.982670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['category'].nunique()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:01:01.986397Z","iopub.execute_input":"2022-07-26T04:01:01.986710Z","iopub.status.idle":"2022-07-26T04:01:01.999538Z","shell.execute_reply.started":"2022-07-26T04:01:01.986681Z","shell.execute_reply":"2022-07-26T04:01:01.998547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(model.classifier.in_features) \nprint(model.classifier.out_features)\n\nfor param in model.parameters():\n    param.requires_grad = False\n    \nn_inputs = model.classifier.in_features\nlast_layer = nn.Linear(n_inputs, N_Classes)\nmodel.classifier = last_layer\nif torch.cuda.is_available():\n    model.cuda()\nprint(model.classifier.out_features)    \n\ncriterion = nn.CrossEntropyLoss()\noptimizer = torch.optim.Adam(model.classifier.parameters())","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:01:02.001163Z","iopub.execute_input":"2022-07-26T04:01:02.001513Z","iopub.status.idle":"2022-07-26T04:01:02.277844Z","shell.execute_reply.started":"2022-07-26T04:01:02.001478Z","shell.execute_reply":"2022-07-26T04:01:02.276495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluation","metadata":{}},{"cell_type":"code","source":"training_history = {'accuracy':[],'loss':[]}\nvalidation_history = {'accuracy':[],'loss':[]}\n\nfrom tqdm import tqdm\n\ndef train(trainloader, model, criterion, optimizer, scaler, device=torch.device(\"cpu\")):\n    train_acc = 0.0\n    train_loss = 0.0\n    for images, labels in tqdm(trainloader):\n        images = images.to(device)\n        labels = labels.to(device)\n        optimizer.zero_grad()\n    with torch.cuda.amp.autocast(enabled=True):\n        output = model(images)\n        loss = criterion(output, labels)\n        scaler.scale(loss).backward()\n        scaler.step(optimizer)\n        scaler.update()\n        acc = ((output.argmax(dim=1) == labels).float().mean())\n        train_acc += acc\n        train_loss += loss\n    return train_acc/len(trainloader), train_loss/len(trainloader)    ","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:01:02.279475Z","iopub.execute_input":"2022-07-26T04:01:02.280012Z","iopub.status.idle":"2022-07-26T04:01:02.289598Z","shell.execute_reply.started":"2022-07-26T04:01:02.279971Z","shell.execute_reply":"2022-07-26T04:01:02.288354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Normal Evaluation\ndef evaluate(testloader, model, criterion, device=torch.device(\"cpu\")):\n    eval_acc = 0.0\n    eval_loss = 0.0\n    for images, labels in tqdm(testloader):\n        images = images.to(device)\n        labels = labels.to(device)\n        with torch.no_grad():\n            output = model(images)\n            loss = criterion(output, labels)\n\n        acc = ((output.argmax(dim=1) == labels).float().mean())\n        eval_acc += acc\n        eval_loss += loss\n  \n    return eval_acc/len(testloader), eval_loss/len(testloader)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:01:02.291352Z","iopub.execute_input":"2022-07-26T04:01:02.292489Z","iopub.status.idle":"2022-07-26T04:01:02.457878Z","shell.execute_reply.started":"2022-07-26T04:01:02.292443Z","shell.execute_reply":"2022-07-26T04:01:02.456710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}