{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train = pd.read_csv('../input/train.csv')\ndf_test = pd.read_csv('../input/test.csv')\ndf_submission = pd.read_csv('../input/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train['diagnosis'].hist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import torch\nimport torch.nn.functional as F\nimport argparse\nimport cv2\nimport numpy as np\nfrom glob import glob\nimport copy\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"num_classes = 5\nimg_height, img_width = 96, 96\nchannel = 3\nGPU = True\ntorch.manual_seed(0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class ResNeXtBlock(torch.nn.Module):\n    def __init__(self, in_f, f_1, out_f, stride=1, cardinality=32):\n        super(ResNeXtBlock, self).__init__()\n\n        self.stride = stride\n        self.fit_dim = False\n        \n        self.block = torch.nn.Sequential(\n            torch.nn.Conv2d(in_f, f_1, kernel_size=1, padding=0, stride=stride),\n            torch.nn.BatchNorm2d(f_1),\n            torch.nn.ReLU(),\n            torch.nn.Conv2d(f_1, f_1, kernel_size=3, padding=1, stride=1, groups=cardinality),\n            torch.nn.BatchNorm2d(f_1),\n            torch.nn.ReLU(),\n            torch.nn.Conv2d(f_1, out_f, kernel_size=1, padding=0, stride=1),\n            torch.nn.BatchNorm2d(out_f),\n            torch.nn.ReLU(),\n        )\n\n        if in_f != out_f:\n            self.fit_conv = torch.nn.Conv2d(in_f, out_f, kernel_size=1, padding=0, stride=1)\n            self.fit_dim = True\n            \n    def forward(self, x):\n        res_x = self.block(x)\n\n        if self.fit_dim:\n            x = self.fit_conv(x)\n\n        if self.stride == 2:\n            x = F.max_pool2d(x, 2, stride=2)\n\n        x = torch.add(res_x, x)\n        x = F.relu(x)\n        \n        return x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class ResNeXt50(torch.nn.Module):\n    def __init__(self):\n        super(ResNeXt50, self).__init__()\n\n        self.conv1 = torch.nn.Conv2d(3, 64, kernel_size=7, padding=3, stride=2)\n        self.bn1 = torch.nn.BatchNorm2d(64)\n        \n        \n        self.block2_1 = ResNeXtBlock(64, 64, 256)\n        self.block2_2 = ResNeXtBlock(256, 64, 256)\n        self.block2_3 = ResNeXtBlock(256, 64, 256)\n\n        self.block3_1 = ResNeXtBlock(256, 128, 512, stride=2)\n        self.block3_2 = ResNeXtBlock(512, 128, 512)\n        self.block3_3 = ResNeXtBlock(512, 128, 512)\n        self.block3_4 = ResNeXtBlock(512, 128, 512)\n\n        self.block4_1 = ResNeXtBlock(512, 256, 1024, stride=2)\n        self.block4_2 = ResNeXtBlock(1024, 256, 1024)\n        self.block4_3 = ResNeXtBlock(1024, 256, 1024)\n        self.block4_4 = ResNeXtBlock(1024, 256, 1024)\n        self.block4_5 = ResNeXtBlock(1024, 256, 1024)\n        self.block4_6 = ResNeXtBlock(1024, 256, 1024)\n\n        self.block5_1 = ResNeXtBlock(1024, 512, 2048, stride=2)\n        self.block5_2 = ResNeXtBlock(2048, 512, 2048)\n        self.block5_3 = ResNeXtBlock(2048, 512, 2048)\n        \n        self.linear = torch.nn.Linear(2048, num_classes)\n        \n        \n    def forward(self, x):\n        x = self.conv1(x)\n        x = self.bn1(x)\n        x = F.relu(x)\n        x = F.max_pool2d(x, 3, padding=1, stride=2)\n\n        x = self.block2_1(x)\n        x = self.block2_2(x)\n        x = self.block2_3(x)\n\n        x = self.block3_1(x)\n        x = self.block3_2(x)\n        x = self.block3_3(x)\n        x = self.block3_4(x)\n\n        x = self.block4_1(x)\n        x = self.block4_2(x)\n        x = self.block4_3(x)\n        x = self.block4_4(x)\n        x = self.block4_5(x)\n        x = self.block4_6(x)\n\n        x = self.block5_1(x)\n        x = self.block5_2(x)\n        x = self.block5_3(x)\n\n        x = F.avg_pool2d(x, [img_height//32, img_width//32], padding=0, stride=1)\n        x = x.view(list(x.size())[0], -1)\n        x = self.linear(x)\n        x = F.softmax(x, dim=1)\n        \n        return x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# get train data\ndef data_load(path, hf=False, vf=False, rot=False):\n    xs = []\n    ts = []\n    paths = []\n    \n    for dir_path in tqdm(glob(path + '/*')):\n        x = cv2.imread(dir_path)\n        x = cv2.resize(x, (img_width, img_height)).astype(np.float32)\n        x /= 255.\n        x = x[..., ::-1]\n        name = dir_path.split(\"/\")[3].split(\".png\")[0]\n        xs.append(x)\n        #print(name)\n        #print(df_train[df_train['id_code']==name]['diagnosis'].values[0])\n        ichi = dir_path.split(\"/\")[2]\n        if 'train_images' == ichi:\n            t = df_train[df_train['id_code']==name]['diagnosis'].values[0]\n        else:\n            t = 0\n        ts.append(t)\n\n        paths.append(dir_path)\n        \n    xs = np.array(xs, dtype=np.float32)\n    ts = np.array(ts, dtype=np.int)\n    \n    xs = xs.transpose(0,3,1,2)\n\n    return xs, ts, paths","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# train\ndef train():\n    # GPU\n    device = torch.device(\"cuda\" if GPU else \"cpu\")\n\n    # model\n    model = ResNeXt50().to(device)\n    opt = torch.optim.SGD(model.parameters(), lr=0.0001, momentum=0.9)\n    model.train()\n\n    xs, ts, paths = data_load('../input/train_images/')\n\n    # training\n    mb = 128\n    mbi = 0\n    train_ind = np.arange(len(xs))\n    np.random.seed(0)\n    np.random.shuffle(train_ind)\n\n    loss_fn = torch.nn.CrossEntropyLoss()\n    \n    for i in range(5000):\n        if mbi + mb > len(xs):\n            mb_ind = copy.copy(train_ind)[mbi:]\n            np.random.shuffle(train_ind)\n            mb_ind = np.hstack((mb_ind, train_ind[:(mb-(len(xs)-mbi))]))\n        else:\n            mb_ind = train_ind[mbi: mbi+mb]\n            mbi += mb\n\n        x = torch.tensor(xs[mb_ind], dtype=torch.float).to(device)\n        t = torch.tensor(ts[mb_ind], dtype=torch.long).to(device)\n\n        opt.zero_grad()\n        y = model(x)\n        #y = F.log_softmax(y, dim=1)\n        loss = loss_fn(y, t)\n        \n        loss.backward()\n        opt.step()\n    \n        pred = y.argmax(dim=1, keepdim=True)\n        acc = pred.eq(t.view_as(pred)).sum().item() / mb\n\n        if (i + 1) % 50 == 0:\n            print(\"iter >>\", i+1, ', loss >>', loss.item(), ', accuracy >>', acc)\n\n    torch.save(model.state_dict(), 'cnn.pt')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# test\ndef test():\n    device = torch.device(\"cuda\" if GPU else \"cpu\")\n    model = ResNeXt50().to(device)\n    model.eval()\n    model.load_state_dict(torch.load('cnn.pt'))\n\n    xs, ts, paths = data_load('../input/test_images/')\n    \n    pred_list =[]\n\n    for i in range(len(paths)):\n        x = xs[i]\n        t = ts[i]\n        path = paths[i]\n        \n        x = np.expand_dims(x, axis=0)\n        x = torch.tensor(x, dtype=torch.float).to(device)\n        \n        pred = model(x)\n        pred = F.softmax(pred, dim=1).detach().cpu().numpy()[0]\n    \n        #print(\"in {}, predicted probabilities >> {}\".format(path, pred))\n        pred_list.append(np.argmax(pred))\n    \n    return pred_list","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pre = test()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_submission['diagnosis'] = pre","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_submission['diagnosis'].hist(bins=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_submission.to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}