{"cells":[{"metadata":{},"cell_type":"markdown","source":"Code has been borrowed from https://www.kaggle.com/abhishek/melanoma-detection-with-pytorch. Thanks to Abhishek!","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"## Import Libraries","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install wtfml==0.0.2","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn import model_selection\nfrom itertools import product\nfrom collections import OrderedDict\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\n\nimport torchvision\nfrom torchvision import datasets, transforms\n\nfrom torch.utils.data import DataLoader\nfrom torch.utils.data.sampler import SubsetRandomSampler\nfrom torch.utils.tensorboard import SummaryWriter\n\nfrom wtfml.data_loaders.image import ClassificationLoader\nimport albumentations\n\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Creating Pytorch Train, Test and Validation Datasets","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# create 5 stratified k-folds in training set\nmelanoma_path=\"../input/siim-isic-melanoma-classification/\"\nmelanoma_image_path=\"../input/siic-isic-224x224-images/\"\n\ndf = pd.read_csv(melanoma_path + \"train.csv\")\ndf[\"kfold\"] = -1    \ndf = df.sample(frac=1).reset_index(drop=True)\ny = df.target.values\nkf = model_selection.StratifiedKFold(n_splits=5)\n\nfor f, (t_, v_) in enumerate(kf.split(X=df, y=y)):\n    df.loc[v_, 'kfold'] = f\n\n# Create train and validation indices\ndf_train = df[df.kfold != 0].reset_index(drop=True)\ndf_valid = df[df.kfold == 0].reset_index(drop=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"training_data_path=melanoma_image_path + \"train/\"\nmean = (0.485, 0.456, 0.406)\nstd = (0.229, 0.224, 0.225)\n\ntrain_images = df_train.image_name.values.tolist()\ntrain_images = [os.path.join(training_data_path, i + \".png\") for i in train_images]\ntrain_targets = df_train.target.values\n\ntrain_aug = albumentations.Compose([\n    albumentations.Normalize(mean, std, max_pixel_value=255.0, always_apply=True),\n#     albumentations.ShiftScaleRotate(shift_limit=0.0625, scale_limit=0.1, rotate_limit=15),\n#     albumentations.Flip(p=0.5)\n])\n\ntrain_dataset = ClassificationLoader(\n    image_paths=train_images,\n    targets=train_targets,\n    resize=None,\n    augmentations=train_aug,\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"valid_images = df_valid.image_name.values.tolist()\nvalid_images = [os.path.join(training_data_path, i + \".png\") for i in valid_images]\nvalid_targets = df_valid.target.values\n\nvalid_aug = albumentations.Compose([\n    albumentations.Normalize(mean, std, max_pixel_value=255.0, always_apply=True)\n])\n\nvalid_dataset = ClassificationLoader(\n    image_paths=valid_images,\n    targets=valid_targets,\n    resize=None,\n    augmentations=valid_aug,\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_data_path = melanoma_image_path + \"test/\"\ndf_test = pd.read_csv(melanoma_path + \"test.csv\")\n\ntest_aug = albumentations.Compose([\n        albumentations.Normalize(mean, std, max_pixel_value=255.0, always_apply=True)\n])\n\ntest_images = df_test.image_name.values.tolist()\ntest_images = [os.path.join(test_data_path, i + \".png\") for i in test_images]\ntest_targets = np.zeros(len(test_images))\n\ntest_dataset = ClassificationLoader(\n    image_paths=test_images,\n    targets=test_targets,\n    resize=None,\n    augmentations=test_aug,\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"## CNN","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"class Net(nn.Module):\n    def __init__(self):\n        super(Net, self).__init__()\n        self.conv1 = nn.Conv2d(in_channels=3, out_channels=6, kernel_size=9, padding=4)\n        self.conv3 = nn.Conv2d(in_channels=6, out_channels=12, kernel_size=7, padding=3)\n        self.conv5 = nn.Conv2d(in_channels=12, out_channels=18, kernel_size=5, padding=2)\n#         self.conv2 = nn.Conv2d(in_channels=6, out_channels=9, kernel_size=7, padding=3)\n#         self.conv4 = nn.Conv2d(in_channels=12, out_channels=15, kernel_size=5, padding=2)\n#         self.conv6 = nn.Conv2d(in_channels=18, out_channels=24, kernel_size=3, padding=1)\n        self.pool1 = nn.MaxPool2d(kernel_size=4, stride=4)\n        self.pool2 = nn.MaxPool2d(kernel_size=2, stride=2)\n        self.fc1 = nn.Linear(in_features=18*7*7, out_features=100)\n        self.fc2 = nn.Linear(in_features=100, out_features=20)\n        self.out = nn.Linear(in_features=20, out_features=2)\n        \n    def forward(self, x):\n        x = F.relu(self.pool1(self.conv1(x)))  # Layer 1\n        x = F.relu(self.pool1(self.conv3(x)))  # Layer 2\n        x = F.relu(self.pool2(self.conv5(x)))  # Layer 3\n        x = x.reshape(-1, 18*7*7)\n        x = F.relu(self.fc1(x))\n        x = F.relu(self.fc2(x))\n        x = self.out(x)\n        \n        return x","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Hyperparameters","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"shuffle=True\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\nprint(device)\n\nparameters = OrderedDict(\n    batch_size=[100],\n    lr = [0.01],\n)\n\nparam_values = [v for v in parameters.values()]\nprint(param_values)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Training","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n\nfor batch_size, lr in product(*param_values):\n    \n    \n    train_loader = torch.utils.data.DataLoader(train_dataset, batch_size=batch_size, shuffle=True, num_workers=2)\n    valid_loader = torch.utils.data.DataLoader(valid_dataset, batch_size=batch_size, shuffle=False, num_workers=2)\n#     trainloader = DataLoader(dataset, batch_size=batch_size, num_workers=2, sampler = train_sampler)\n#     testloader = DataLoader(dataset, batch_size=batch_size, num_workers=2, sampler = valid_sampler)\n\n    net = Net().to(device)\n    optimizer = optim.Adam(net.parameters(), lr=lr)\n    criterion = nn.CrossEntropyLoss()\n\n    comment = f'melanoma batch_size={batch_size} lr={lr}'\n    print(comment)\n#     tb = SummaryWriter(comment=comment)\n#     tb_count=0\n\n    for epoch in range(6): \n        running_loss = 0.0\n        for i, data in enumerate(train_loader, 0):\n            inputs, labels = data['image'].to(device), data['targets'].to(device)\n            optimizer.zero_grad()\n            outputs = net(inputs)\n            loss = criterion(outputs, labels)\n            loss.backward()\n            optimizer.step()\n\n            running_loss += loss.item()\n            if i % 50 == 49:   \n#                 tb_count += 1\n#                 tb.add_scalar('Running Loss', running_loss/100, tb_count)\n                print('[%d, %5d] loss: %.5f' %(epoch + 1, i + 1, running_loss / 50))\n                running_loss = 0.0\n\n        if epoch % 2 == 1:\n            print('At the end of epoch %d' %(epoch+1))\n            correct = 0\n            total = 0\n            with torch.no_grad():\n                preds=[]\n                targets=[]\n                for data in train_loader:\n                    images, labels = data['image'].to(device), data['targets'].to(device)\n                    outputs = net(images)\n                    _, predicted = torch.max(outputs.data, 1)\n                    preds += list(predicted.cpu().detach().numpy().squeeze())\n                    targets += list(labels.cpu().detach().numpy().squeeze())\n                    total += labels.size(0)\n                    correct += (predicted == labels).sum().item()\n\n    #             tb.add_scalar('Train Accuracy', 100 * correct / total, epoch+1)\n            print('Accuracy of the network on the train images: %d %%' % (100 * correct / total))\n            print('Training Confusion Matrix:')\n            print(confusion_matrix(targets, preds))\n\n            with torch.no_grad():\n                for data in valid_loader:\n                    images, labels = data['image'].to(device), data['targets'].to(device)\n                    outputs = net(images)\n                    _, predicted = torch.max(outputs.data, 1)\n                    preds += list(predicted.cpu().detach().numpy().squeeze())\n                    targets += list(labels.cpu().detach().numpy().squeeze())\n                    total += labels.size(0)\n                    correct += (predicted == labels).sum().item()\n    #             tb.add_scalar('Test Accuracy', 100 * correct / total, epoch+1)\n            print('Accuracy of the network on the validation images: %d %%' % (100 * correct / total))\n            print('Validation Confusion Matrix:')\n            print(confusion_matrix(targets, preds))\n\n#     tb.close()\n    print('Finished Training')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_loader = torch.utils.data.DataLoader(test_dataset, batch_size=200, shuffle=False, num_workers=4)\nwith torch.no_grad():\n    preds=[]\n    targets=[]\n    for data in train_loader:\n        images, labels = data['image'].to(device), data['targets'].to(device)\n        outputs = net(images)\n        _, predicted = torch.max(outputs.data, 1)\n        preds += list(predicted.cpu().detach().numpy().squeeze())\n        targets += list(labels.cpu().detach().numpy().squeeze())\n        total += labels.size(0)\n        correct += (predicted == labels).sum().item()\n\n# tb.add_scalar('Train Accuracy', 100 * correct / total, epoch+1)\nprint('Accuracy of the network on the train images: %d %%' % (100 * correct / total))\nprint(len(preds), len(targets))\nprint('Training Confusion Matrix:')\nprint(confusion_matrix(targets, preds))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}