{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\nimport torch\nimport torchvision\nfrom torch import nn\nfrom torch import optim\nfrom torch.autograd import Variable\nimport torch.nn.functional as F\nfrom torchvision import datasets, transforms, models\nfrom torch.utils.data import DataLoader, Dataset\nimport torch.utils.data as utils\nprint(\"PyTorch Version: \",torch.__version__)\nprint(\"Torchvision Version: \",torchvision.__version__)\n\nfrom PIL import Image\nimport cv2\n\nfrom sklearn.preprocessing import LabelEncoder, OneHotEncoder\nfrom sklearn.model_selection import train_test_split\n\nimport time\nimport copy\nimport glob\nimport sys\nsys.setrecursionlimit(100000)  # To increase the capacity of the stack\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# check if CUDA is available\ntrain_on_gpu = torch.cuda.is_available()\n\nfrom os import path\nfrom wheel.pep425tags import get_abbr_impl, get_impl_ver, get_abi_tag\nplatform = '{}{}-{}'.format(get_abbr_impl(), get_impl_ver(), get_abi_tag())\ncuda_output = !ldconfig -p|grep cudart.so|sed -e 's/.*\\.\\([0-9]*\\)\\.\\([0-9]*\\)$/cu\\1\\2/'\naccelerator = cuda_output[0]\n\nif not train_on_gpu:\n    print('CUDA is not available. Training on CPU ...')\nelse:\n    print('CUDA is available! Training on GPU ...')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Load data\ntrain_dir = '../input/aptos2019-blindness-detection/train_images/'\n\ntrain = pd.read_csv('../input/aptos2019-blindness-detection/train.csv')\ntest = pd.read_csv('../input/aptos2019-blindness-detection/test.csv')\n\nsample_submission = pd.read_csv('../input/aptos2019-blindness-detection/sample_submission.csv')\n\n# Split off data for validation set\ntrain, valid = train_test_split(train, train_size=0.75, test_size=0.25, shuffle=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Number of train samples: ', train.shape[0])\nprint('Number of validation samples: ', valid.shape[0])\nprint('Number of test samples: ', test.shape[0])\ndisplay(train.head(10))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.DataFrame(train)\n\nClassCounts = pd.value_counts(df['diagnosis'], sort=True)\nprint(ClassCounts)\nplt.figure(figsize=(10, 7))\nClassCounts.plot.bar(rot=0);\nplt.title('Severity Counts for Training Data');","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# To display 5 unique retina images from each of the 5 classes\n\nj = 1\nfig=plt.figure(figsize=(15, 15))\nfor class_id in sorted(train['diagnosis'].unique()):\n    plot_no = j\n    for i, (idx, row) in enumerate(train.loc[train['diagnosis'] == class_id].sample(5).iterrows()):\n        ax = fig.add_subplot(5, 5, plot_no)\n        im = Image.open(f\"../input/aptos2019-blindness-detection/train_images/{row['id_code']}.png\")\n        plt.imshow(im)\n        ax.set_title(f'Label: {class_id}')\n        plot_no += 5\n    j += 1\n\nplt.show()\nplt.savefig(\"samples_viz.png\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import PIL\n\nclass ImageLoader(Dataset):\n    \n    def __init__(self, df, datatype):\n        self.datatype = datatype\n        #self.labels = df['diagnosis'].values\n        if self.datatype == 'train':\n            self.image_files = [f'../input/aptos2019-blindness-detection/train_images/{i}.png' for i in train['id_code'].values]\n            self.transform = transforms.Compose([\n                                                 transforms.RandomVerticalFlip(p=0.5),\n                                                 transforms.RandomHorizontalFlip(p=0.5),\n                                                 #transforms.Grayscale(num_output_channels=3),\n                                                 transforms.ToTensor(),\n                                                 transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225])\n                                                ])\n            self.labels = train['diagnosis'].values\n        elif self.datatype == 'valid':\n            self.image_files = [f'../input/aptos2019-blindness-detection/train_images/{i}.png' for i in valid['id_code'].values]\n            self.transform = transforms.Compose([\n                                                #transforms.Grayscale(num_output_channels=3),\n                                                transforms.RandomVerticalFlip(p=0.5),\n                                                transforms.RandomHorizontalFlip(p=0.5),\n                                                transforms.ToTensor(),\n                                                transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225])\n                                                ])\n            self.labels = valid['diagnosis'].values\n        \n    def __len__(self):\n        return len(self.image_files)\n\n    def __getitem__(self, index):\n        image = Image.open(self.image_files[index])\n        image = image.convert('RGB')\n        image = image.resize((224, 224))\n        #image = PIL.ImageOps.autocontrast(image)\n        image = self.transform(image)\n        if self.datatype == 'train':\n            label = torch.tensor(self.labels[index], dtype=torch.long)\n            return image, label\n        elif self.datatype == 'valid':\n            label = torch.tensor(self.labels[index], dtype=torch.long)\n            return image, label","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"retinaImages = ImageLoader(df=valid, datatype='valid')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(retinaImages)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"valid['diagnosis'][13+len(train)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"type(retinaImages[13][0])\nprint(retinaImages[13][0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.imshow(retinaImages[13][0].permute(1, 2, 0))\nprint(\"Label: \" + str(retinaImages[13][1]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"trainloader = torch.utils.data.DataLoader(ImageLoader(df=train, datatype='train'), batch_size=60, shuffle=True)\ntestloader = torch.utils.data.DataLoader(ImageLoader(df=valid, datatype='valid'), batch_size=60, shuffle=False)  # serving as validation set...","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#model = models.densenet121(pretrained=False)\nmodel = models.resnet50(pretrained=True)\nmodel\n\nif train_on_gpu:\n    model = model.cuda()\nmodel","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cuda_output[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_checkpoint(filepath):\n    checkpoint = torch.load(filepath)\n    model = checkpoint['model']\n    model.load_state_dict(checkpoint['state_dict'])\n    for parameter in model.parameters():\n        parameter.requires_grad = False\n\n    model.eval()\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Freeze parameters so we don't backprop through them\nfor param in model.parameters():\n    param.requires_grad = False\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\") \n\nfrom collections import OrderedDict\nfc = nn.Sequential(OrderedDict([\n                          ('fc1', nn.Linear(2048, 1024)),\n                          ('relu', nn.ReLU()),\n                          ('dropout', nn.Dropout(p=0.2)),\n                          ('fc2', nn.Linear(1024, 5)),\n                          ('output', nn.LogSoftmax(dim=1))\n                          ]))\n  \nweights = torch.tensor([2., 11., 5., 13., 12.])\nweights = weights.to(device)\ncriterion = nn.NLLLoss(weight=weights, reduction='mean')\n#criterion = nn.NLLLoss()\n\n# Only train the classifier parameters, feature parameters are frozen\noptimizer = optim.SGD(model.fc.parameters(), lr=0.0005, momentum=0.9)\n\n#model = load_checkpoint('/kaggle/checkpoint.pth')\n\nmodel.fc = nn.Linear(512, 5)\nmodel.fc = fc\n\nmodel.to(device)    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"checkpoint = {'model': fc,\n          'state_dict': model.state_dict(),\n          'optimizer' : optimizer.state_dict()}\n\ntorch.save(checkpoint, '/kaggle/checkpoint.pth')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!ls /kaggle/working","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i, (inputs, labels) in enumerate(trainloader):\n    # Move input and label tensors to the GPU\n    \n    inputs, labels = inputs.to(device), labels.to(device)\n    \n    start = time.time()\n\n    outputs = model.forward(inputs)\n    loss = criterion(outputs, labels)\n    loss.backward()\n    optimizer.step()\n\n    if i==3:\n        break\n        \nprint(f\"Device = {device}; Time per batch: {(time.time() - start)/3:.3f} seconds\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"epochs = 5\nsteps = 0\nrunning_loss = 0\nprint_every = 10\n\nvalidation_accuracy = []\n\nfor epoch in range(epochs):\n    for inputs, labels in trainloader:\n        steps += 1\n        # Move input and label tensors to the default device\n        inputs, labels = inputs.to(device), labels.to(device)\n        \n        optimizer.zero_grad()\n        \n        logps = model.forward(inputs)\n        loss = criterion(logps, labels)\n        loss.backward()\n        optimizer.step()\n\n        running_loss += loss.item()\n        \n        if steps % print_every == 0:\n            test_loss = 0\n            accuracy = 0\n            model.eval()\n            \n            with torch.no_grad():\n                for inputs, labels in testloader:\n                    inputs, labels = inputs.to(device), labels.to(device)\n                    logps = model.forward(inputs)\n                    #print(logps)\n                    batch_loss = criterion(logps, labels)\n                    \n                    test_loss += batch_loss.item()\n                    \n                    # Calculate accuracy\n                    ps = torch.exp(logps)\n                    #print(ps)\n                    top_p, top_class = ps.topk(1, dim=1)\n                    #print(top_class)\n                    equals = top_class == labels.view(*top_class.shape)\n                    \n                    accuracy += torch.mean(equals.type(torch.FloatTensor)).item()\n                    \n                    validation_accuracy.append(accuracy)\n            \n            print(f\"Epoch {epoch+1}/{epochs}... \"\n                  f\"Validation accuracy: {accuracy/len(testloader):.3f}\"\n                  )\n            \n        running_loss = 0\n        model.train()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%matplotlib inline\n%config InlineBackend.figure_format = 'retina'\n\nimport matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.plot(validation_accuracy, label='Validation accuracy')\nplt.legend(frameon=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(test['id_code'].values)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class SubmissionLoader(Dataset):\n    \n    def __init__(self, df):\n        self.datatype = 'test'\n        self.image_files = [f'../input/aptos2019-blindness-detection/test_images/{i}.png' for i in test['id_code'].values]\n        self.transform = transforms.Compose([\n                                            #transforms.Grayscale(num_output_channels=3),\n                                            transforms.ToTensor(),\n                                            transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225])\n                                            ])\n        self.id_code = test['id_code'].values\n\n    def __len__(self):\n        return len(self.image_files)\n\n    def __getitem__(self, index):\n        image = Image.open(self.image_files[index])\n        image = image.convert('RGB')\n        image = image.resize((224, 224))\n        #image = PIL.ImageOps.autocontrast(image)\n        image = self.transform(image)\n        id_code = self.id_code[index]\n        return image, id_code","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submissions = torch.utils.data.DataLoader(SubmissionLoader(df=test), batch_size=1, shuffle=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(submissions)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds = []\nid_codes = []\n\nmodel.eval()\nwith torch.no_grad():\n\n    for i, (image, id_code) in enumerate(submissions):\n\n        image = image.to(device)\n        output = model.forward(image)\n        ps = torch.exp(output)\n        top_p, top_class = ps.topk(1, dim=1)\n        pred = torch.squeeze(top_class).item()\n        preds.append(pred)\n        id_codes.append(id_code)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"output = pd.read_csv(\"../input/aptos2019-blindness-detection/sample_submission.csv\")\npreds = list(map(int, preds))\noutput.diagnosis = preds\noutput.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pd.read_csv('/kaggle/working/submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"freq, _ = np.histogram(output.diagnosis, density=True, bins=5)\nfreq","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}