{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import os\nimport torch\nimport cv2\n#import timm\nimport IPython\n\nimport albumentations as albu\nimport numpy as np\nimport pandas as pd\n#import efficientnet_pytorch as enet\n\nfrom torch.optim import SGD, Adam\n#from minetorch import Miner\n#from minetorch.metrics import MultiClassesClassificationMetricWithLogic\n#from ranger_adabelief import RangerAdaBelief\nfrom albumentations import OneOf, Compose\nfrom albumentations.pytorch import ToTensorV2\nfrom torch.utils.data import DataLoader, Dataset\nfrom sklearn.model_selection import train_test_split\nfrom tqdm.notebook import tqdm\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom torch.optim.lr_scheduler import StepLR\nfrom torch.utils.data import DataLoader, Dataset\nfrom torchvision import datasets, transforms\n\n\n\n\nbs = 32\nimage_size = 384\nlearning_rate = 1e-3\n\nanno = '/kaggle/input/cassava-leaf-disease-classification/train.csv'\nsample = '/kaggle/input/cassava-leaf-disease-classification/sample_submission.csv'\n\ndf = pd.read_csv(anno)\ndf_sample = pd.read_csv(sample)\ndf.head(2)\n\n#TRAIN = '/home/featurize/data/train_images'\n#TEST = '/home/featurize/data/test_images'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!git clone https://github.com/DingXiaoH/RepVGG.git","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Defining the paths to files.\ntrainimages = \"/kaggle/input/cassava-leaf-disease-classification/train_images\"\n\ntraindf = pd.read_csv(\"/kaggle/input/cassava-leaf-disease-classification/train.csv\")\n\ntrain_list = []\n# Converting the Image IDs to their paths.\nfor i in traindf.index:\n    \n    a = traindf[\"image_id\"].loc[i]\n    \n    b = trainimages + \"/\" + a\n    \n    train_list.append((b, traindf['label'].loc[i]))\n\n# Taking a look.\ntrain_list[:3]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from torch.utils.data import DataLoader, Dataset\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class CLDCdataset(Dataset):\n    def __init__(self, df, train_dir, transforms):\n        self.train_dir = train_dir\n        self.df = df\n        self.transforms = transforms\n    \n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        images = []\n        image = cv2.imread(os.path.join(self.train_dir, row['image_id']))\n        img = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        augmented = self.transforms(image=img)\n        label = row['label']\n        return augmented['image'], label\n\n    def __len__(self):\n        return len(self.df)\n    \n\ndef make_transforms(phase, size, mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)):\n    list_transforms = []\n    if phase == \"train\":\n         list_transforms.extend(\n            [\n                OneOf([\n                    albu.Resize(size, size),\n                    albu.Transpose(p=1),\n                    albu.VerticalFlip(p=1),\n                    albu.HorizontalFlip(p=1)\n                ], p=0.5),\n                OneOf([\n                    albu.RandomCrop(width=300, height=300, p=1),\n                    albu.ShiftScaleRotate(p=1)\n                ],p=0.5),\n            ]\n         )\n    list_transforms.extend(\n        [\n            albu.Resize(size, size),\n            albu.Normalize(mean=mean, std=std, p=1),\n            ToTensorV2(),\n        ]\n    )\n    list_trfms = Compose(list_transforms)\n    return list_trfms\n\nX_train, X_test, y_train, y_test = train_test_split(\n    df['image_id'], \n    df['label'], \n    test_size=0.2, \n    random_state=99, \n    stratify=df['label']\n)\n\ntrain_df = df.loc[X_train.index]\nval_df = df.loc[X_test.index]\nprint(train_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('训练集: ')\nprint('==================')\nfor i in range(5):\n    print(f'第{i}类样本数量:', len(train_df[train_df.label == i]))\n\n\nprint('测试集: ')\nprint('==================')\nfor i in range(5):\n    print(f'第{i}类样本数量:', len(val_df[val_df.label == i]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nTRAIN = '/kaggle/input/cassava-leaf-disease-classification/train_images/'\nTEST = '/kaggle/input/cassava-leaf-disease-classification/test_images/'\n\n\ntrain_dataset = CLDCdataset(train_df, TRAIN, make_transforms('train', image_size))\nval_dataset = CLDCdataset(val_df, TRAIN, make_transforms('val', image_size))\nprint(\"train_dataset: \", train_dataset)\n# Get dataloader for pytorch to train\ntrain_loader = torch.utils.data.DataLoader(\n    train_dataset, batch_size=bs, shuffle=True)\nvalid_loader = torch.utils.data.DataLoader(\n    val_dataset, batch_size=bs, shuffle=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cd RepVGG/\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from repvgg import repvgg_model_convert, create_RepVGG_A0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"seed = 42\nimport random\n\ndef seed_it(seed):\n    random.seed(seed)\n    os.environ[\"PYTHONSEED\"] = str(seed)\n    np.random.seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = True\n    \n# Running the Function.\nseed_it(seed)\ndevice = torch.device(\"cuda:0\")\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"batch_size = 64\nepochs = 30\nlr = 3e-4\ngamma= 0.7\nseed = 50","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nmodel = create_RepVGG_A0(deploy=False).to(device)\n\ncriterion = nn.CrossEntropyLoss()\n\noptimizer = optim.Adam(model.parameters(), lr = lr)\n\nscheduler = StepLR(optimizer, step_size=1, gamma=gamma)\n\n\npre_save = 0\nfor epoch in range(epochs):\n    epoch_loss = 0.0\n    epoch_accuracy = 0.0\n    \n    for data, label in tqdm(train_loader):\n        data = data.to(device)\n        label = label.to(device)\n        model.to(device)\n        output = model(data)\n        \n        label = torch.nn.functional.one_hot(label, num_classes = 5)\n        label = label.squeeze_()\n        label = torch.argmax(label, axis=1)\n        \n        label = label.type_as(output)\n        \n        loss = criterion(output, label.long())\n        \n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n        \n        acc = (output.argmax(dim=1) == label).float().mean()\n        \n        epoch_accuracy += acc / len(train_loader)\n        epoch_loss += loss / len(train_loader)\n        \n    with torch.no_grad():\n        epoch_val_accuracy = 0\n        epoch_val_loss = 0        \n        \n        for data, label in valid_loader:\n            data = data.to(device)\n            model.to(device)\n            \n            label = label.to(device)\n            #label = label.squeeze_()\n            label = label.type_as(output)\n            \n            val_output = model(data)\n            val_loss = criterion(val_output, label.long())\n            \n            accc = (val_output.argmax(dim = 1) == label).float().mean()\n            epoch_val_accuracy += acc / len(valid_loader)\n            epoch_val_loss += val_loss / len(valid_loader)\n            \n            \n    print(\n        f\"Epoch: {epoch+1} - loss: {epoch_loss:.4f} - acc : {epoch_accuracy:.4f} - val_loss : {epoch_val_loss:.4f} - val_acc: {epoch_val_accuracy: .4f}\\n\"\n    )\n    if epoch_val_accuracy > pre_save:\n        print(\"save epoch:\", epoch + 1)\n        print(\"epoch_val_accuracy:\", epoch_val_accuracy)\n        pre_save = epoch_val_accuracy\n        torch.save(model.state_dict(), './checkpoint.pth')\n    \n    \n    \n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}