{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import cv2\nimport torch\nimport numpy as np\nimport torch.nn as nn\nfrom torch.utils.data import Dataset,DataLoader\nfrom torchvision import transforms,models\n#from transform import get_train_transform\nimport pandas as pd\nimport torch.nn.functional as F\n\n# 判定GPU是否存在\n# device = torch.device('cuda:0' if torch.cuda.is_available() else 'cpu')\n#GPU是否存在\n# device = torch.device(\"cuda:0\")\ndevice = torch.device('cuda:0' if torch.cuda.is_available() else 'cpu')\n# device = 'cuda:0'\nprint ('device:', device)\n\nINPUT_PATH = \"/kaggle/input/bengaliai-cv19/\"\nCKPT_path = '/kaggle/input/' + '/ckpt11/resnet50_enc_4.ckpt'\n\nHEIGHT = 137\nWIDTH = 236\nTARGET_SIZE = 256\n# 定义超参数\nbatch_size = 64\nSIZE = 128\n\n# 中心填充\ndef make_square(img, target_size=256):\n    img = img[0:-1, :]\n    height, width = img.shape\n\n    x = target_size\n    y = target_size\n\n    square = np.ones((x, y), np.uint8) * 255\n    square[(y - height) // 2:y - (y - height) // 2, (x - width) // 2:x - (x - width) // 2] = img\n\n    return square\n\n\n# 模型 \nimport cv2\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset,DataLoader\nfrom torchvision import transforms,models\n\n\nclass ResidualBlock(nn.Module):\n    def __init__(self,in_channels,out_channels,stride=1,kernel_size=3,padding=1,bias=False):\n        super(ResidualBlock,self).__init__()\n        self.cnn1 =nn.Sequential(\n            nn.Conv2d(in_channels,out_channels,kernel_size,stride,padding,bias=False),\n            nn.BatchNorm2d(out_channels),\n            nn.ReLU(True)\n        )\n        self.cnn2 = nn.Sequential(\n            nn.Conv2d(out_channels,out_channels,kernel_size,1,padding,bias=False),\n            nn.BatchNorm2d(out_channels)\n        )\n        if stride != 1 or in_channels != out_channels:\n            self.shortcut = nn.Sequential(\n                nn.Conv2d(in_channels,out_channels,kernel_size=1,stride=stride,bias=False),\n                nn.BatchNorm2d(out_channels)\n            )\n        else:\n            self.shortcut = nn.Sequential()\n            \n    def forward(self,x):\n        residual = x\n        x = self.cnn1(x)\n        x = self.cnn2(x)\n        x += self.shortcut(residual)\n        x = nn.ReLU(True)(x)\n        return x\n\nclass ResNet18(nn.Module):\n    def __init__(self):\n        super(ResNet18,self).__init__()\n        \n        self.block1 = nn.Sequential(\n            nn.Conv2d(3,64,kernel_size=2,stride=2,padding=3,bias=False),\n            nn.BatchNorm2d(64),\n            nn.ReLU(True)\n        )\n        \n        self.block2 = nn.Sequential(\n            nn.MaxPool2d(1,1),\n            ResidualBlock(64,64),\n            ResidualBlock(64,64,2)\n        )\n        \n        self.block3 = nn.Sequential(\n            ResidualBlock(64,128),\n            ResidualBlock(128,128,2)\n        )\n        \n        self.block4 = nn.Sequential(\n            ResidualBlock(128,256),\n            ResidualBlock(256,256,2)\n        )\n        self.block5 = nn.Sequential(\n            ResidualBlock(256,512),\n            ResidualBlock(512,512,2)\n        )\n        \n        self.avgpool = nn.AvgPool2d(2)\n        # vowel_diacritic\n        self.fc1 = nn.Linear(8192,11)\n        # grapheme_root\n        self.fc2 = nn.Linear(8192,168)\n        # consonant_diacritic\n        self.fc3 = nn.Linear(8192,7)\n        \n    def forward(self,x):\n        x = self.block1(x)\n        x = self.block2(x)\n        x = self.block3(x)\n        x = self.block4(x)\n        x = self.block5(x)\n        x = self.avgpool(x)\n        x = x.view(x.size(0),-1)\n        x1 = self.fc1(x)\n        x2 = self.fc2(x)\n        x3 = self.fc3(x)\n        return x1,x2,x3\n# 等价于删除某个层\nclass Identity(torch.nn.Module):\n    def __init__(self):\n        super(Identity, self).__init__()\n        \n    def forward(self, x):\n        return x\n\n\ndef resnet50():\n    res50 = models.resnet50(pretrained=False)\n    res50.fc = Identity()  # 最后一层全连接删除\n    res50.avgpool = Identity() # 全局平均池化也需要更改，原始的224降低32x，现在32降低32x成了1\n    return res50\n    \n    \nclass ResNet50(torch.nn.Module):\n    def __init__(self,):\n        super(ResNet50, self).__init__()\n        self.resnet50 = resnet50()\n        self.avgpool = nn.AvgPool2d(2)\n        # vowel_diacritic\n        self.fc1 = nn.Linear(32768,11)\n        # grapheme_root\n        self.fc2 = nn.Linear(32768,168)\n        # consonant_diacritic\n        self.fc3 = nn.Linear(32768,7)\n        \n    def forward(self, x):\n        x = self.resnet50(x)\n        # x = self.avgpool(x)\n        x = x.view(x.size(0),-1)\n        x1 = self.fc1(x)\n        x2 = self.fc2(x)\n        x3 = self.fc3(x)\n        return x1,x2,x3\n\n# 数据增强\n#!usr/bin/env python  \n#-*- coding:utf-8 _*- \nimport random\nimport math\nimport torch\n\nfrom PIL import Image, ImageOps, ImageFilter\nfrom torchvision import transforms\n\n\ndef update_lr(optimizer, lr):    \n    for param_group in optimizer.param_groups:\n        param_group['lr'] = lr\n\nclass Resize(object):\n    def __init__(self, size, interpolation=Image.BILINEAR):\n        self.size = size\n        self.interpolation = interpolation\n\n    def __call__(self, img):\n        # padding\n        ratio = self.size[0] / self.size[1]\n        w, h = img.size\n        if w / h < ratio:\n            t = int(h * ratio)\n            w_padding = (t - w) // 2\n            img = img.crop((-w_padding, 0, w+w_padding, h))\n        else:\n            t = int(w / ratio)\n            h_padding = (t - h) // 2\n            img = img.crop((0, -h_padding, w, h+h_padding))\n\n        img = img.resize(self.size, self.interpolation)\n\n        return img\n\nclass RandomRotate(object):\n    def __init__(self, degree, p=0.5):\n        self.degree = degree\n        self.p = p\n\n    def __call__(self, img):\n        if random.random() < self.p:\n            rotate_degree = random.uniform(-1*self.degree, self.degree)\n            img = img.rotate(rotate_degree, Image.BILINEAR)\n        return img\n\nclass RandomGaussianBlur(object):\n    def __init__(self, p=0.5):\n        self.p = p\n    def __call__(self, img):\n        if random.random() < self.p:\n            img = img.filter(ImageFilter.GaussianBlur(\n                radius=random.random()))\n        return img\n\n\ndef get_train_transform(size):\n    train_transform = transforms.Compose([\n        #Resize((int(size * (256 / 224)), int(size * (256 / 224)))),\n        transforms.RandomCrop(size),\n        #transforms.RandomHorizontalFlip(),\n        RandomRotate(15, 0.3),\n        RandomGaussianBlur(),\n        transforms.ToTensor(),\n        # transforms.Normalize(mean=mean, std=std),\n    ])\n    return train_transform\n\ndef get_test_transform(size):\n    return transforms.Compose([\n        #Resize((int(size * (256 / 224)), int(size * (256 / 224)))),\n        transforms.CenterCrop(size),\n        transforms.ToTensor(),\n        # transforms.Normalize(mean=mean, std=std),\n    ])\n\n\nfrom PIL import Image\nimport cv2\n# 中心填充\ndef default_loader(path):      \n    # 注意要保证每个batch的tensor大小时候一样的。      \n    return Image.open(path).convert('RGB')\n    #return cv2.imread(path)\n\n\n# 数据加载模块\nclass BengaliParquetDatasetTest(Dataset):\n    def __init__(self, parquet_file, transform=None, _type=\"train\"):\n        self.data = pd.read_parquet(parquet_file)\n        self.transform = transform\n        self.type = _type\n\n    def __len__(self):\n        return len(self.data)\n\n    def __getitem__(self, idx):\n        if self.type == \"train\":\n            return None\n        \n        if self.type == \"test\":\n            tmp = self.data.iloc[idx, 1:].values.reshape(HEIGHT, WIDTH)\n            img = np.zeros((TARGET_SIZE, TARGET_SIZE, 3))\n            img[..., 0] = make_square(tmp, target_size=TARGET_SIZE)\n            img[..., 1] = img[..., 0]\n            img[..., 2] = img[..., 0]\n            image_id = self.data.iloc[idx, 0]\n            print ('image_id:', image_id)\n            if self.transform:\n                img = Image.fromarray(img.astype('uint8')).resize((130, 130))\n                img = self.transform(img)\n            return image_id, img\n\n\nsubmission_df = pd.read_csv(INPUT_PATH + '/sample_submission.csv')\n# device = torch.device('cuda:0' if torch.cuda.is_available() else 'cpu')\nmodel = ResNet50().to(device)\n# model.load_state_dict(torch.load( CKPT_path, map_location=torch.device('cuda:0') ))\nmodel.load_state_dict(torch.load( CKPT_path, map_location=torch.device('cpu') ))\nmodel.to(device)\n\ntransform_test = get_test_transform( SIZE )\n\n\nresults = []\nresults_id = []\nfor i in range(4):\n    parquet_file = INPUT_PATH + '/test_image_data_{}.parquet'.format(i)\n    print (\"parq:\", parquet_file)\n    test_dataset = BengaliParquetDatasetTest(parquet_file=parquet_file, transform=transform_test, _type=\"test\")\n    data_loader_test = torch.utils.data.DataLoader(test_dataset, batch_size=batch_size,\n                                                   num_workers=12, shuffle=False)\n    \n\n    print('Parquet {}'.format(i))\n    model.eval()\n    tk0 = data_loader_test\n\n    for step, (image_id, images) in enumerate(tk0):\n        inputs = images\n        image_ids = image_id\n        inputs = inputs.to(device, dtype=torch.float)\n\n        out_vowel, out_graph, out_conso = model(inputs)\n        out_vowel = F.softmax(out_vowel, dim=1).data.cpu().numpy().argmax(axis=1)\n        out_graph = F.softmax(out_graph, dim=1).data.cpu().numpy().argmax(axis=1)\n        out_conso = F.softmax(out_conso, dim=1).data.cpu().numpy().argmax(axis=1)\n\n        for idx, image_id in enumerate(image_ids):\n            results.append(out_conso[idx])\n            results.append(out_graph[idx])\n            results.append(out_vowel[idx])\n            \n            results_id.append( str(image_id) + '_consonant_diacritic' ) \n            results_id.append( str(image_id) + '_grapheme_root' )\n            results_id.append( str(image_id) + '_vowel_diacritic' )\n\nsubmission_df['target'] = results\nsubmission_df['row_id'] = submission_df['row_id']\nsubmission_df.to_csv('submission.csv', index=False)\n\nprint ('end!!!!!!!!!')\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_df","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}