{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        #print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-13T07:33:27.540758Z","iopub.execute_input":"2021-07-13T07:33:27.541156Z","iopub.status.idle":"2021-07-13T07:33:27.547797Z","shell.execute_reply.started":"2021-07-13T07:33:27.541117Z","shell.execute_reply":"2021-07-13T07:33:27.546561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n## taking a look at the study level data \ndf_TSL = pd.read_csv('../input/siim-covid19-detection/train_study_level.csv')\n\n#dividing the column name into two\ndf_TSL[[\"StudyInstanceUID\",\"id\"]] = df_TSL[\"id\"].str.split(\"_\", expand = True)\ndf_TSL.drop([\"id\"],axis = 1, inplace = True)\n\ndf_TIL = pd.read_csv('../input/siim-covid19-detection/train_image_level.csv')\n\ndf_TIL[[\"SOP_id\", \"id\"]] = df_TIL.id.str.split(\"_\", expand = True,)\ndf_TIL.drop([\"id\"],inplace =True, axis =1)\n\n#join the two tables\n#df_TIL.join(df_TSL, on=\"StudyInstanceUID\")\ndf_TIL_TSL = pd.merge(\n    df_TIL, \n    df_TSL, \n    left_on=\"StudyInstanceUID\", \n    right_on=\"StudyInstanceUID\", \n    how=\"left\", \n    sort=False\n)\n\ndf_TIL_TSL.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-16T06:00:34.137905Z","iopub.execute_input":"2021-07-16T06:00:34.138249Z","iopub.status.idle":"2021-07-16T06:00:34.486604Z","shell.execute_reply.started":"2021-07-16T06:00:34.138205Z","shell.execute_reply":"2021-07-16T06:00:34.485644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Remove duplicates","metadata":{}},{"cell_type":"code","source":"print(len(df_TIL_TSL))\ndf_SOP_noduplicate = df_TIL_TSL[df_TIL_TSL['StudyInstanceUID'].map((df_TIL_TSL['StudyInstanceUID'].value_counts()) > 1) & (df_TIL_TSL['Negative for Pneumonia'] != 1)].dropna()\nl_duplicates = list(df_SOP_noduplicate.StudyInstanceUID.value_counts().keys())\ndf = df_TIL_TSL[~df_TIL_TSL['StudyInstanceUID'].isin(l_duplicates)]\ndf = df.append(df_SOP_noduplicate)\n\nd=pd.DataFrame()\nfor name, data in df.groupby([\"StudyInstanceUID\"]):\n    if len(data)>1:\n        #print(len(data))\n        df = df[df.StudyInstanceUID != name]\n        d = d.append(data.sample())\n\n\ndf_final = df.append(d)\nprint(len(df_final))","metadata":{"execution":{"iopub.status.busy":"2021-07-16T06:00:37.084529Z","iopub.execute_input":"2021-07-16T06:00:37.084934Z","iopub.status.idle":"2021-07-16T06:00:37.487351Z","shell.execute_reply.started":"2021-07-16T06:00:37.084904Z","shell.execute_reply":"2021-07-16T06:00:37.486378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# creating the labels\n","metadata":{}},{"cell_type":"code","source":"#df_final.loc[df_final.boxes.isnull(),'Label_custom'] = 0\n#df_final.loc[df_final.boxes.notnull(),'Label_custom'] = 1\n\ndef label_race (row):\n    if row['Negative for Pneumonia'] == 1 :\n        return 0\n    if row['Typical Appearance'] == 1 :\n        return 1\n    if row['Indeterminate Appearance'] == 1 :\n        return 2\n    if row['Atypical Appearance'] == 1 :\n        return 3\n   \n    return 'Other'\n\ndf_final['label'] = df_final.apply (lambda row: label_race(row), axis=1)\ndf_final.drop([\"Negative for Pneumonia\",\"Typical Appearance\",\"Indeterminate Appearance\",\"Atypical Appearance\"], axis =1, inplace = True)\n#df_final[\"label\"] = df_final[\"label\"].astype(\"category\")\ndf_final['img_id'] = df_final['SOP_id'].astype('str') +'.jpg'","metadata":{"execution":{"iopub.status.busy":"2021-07-16T06:00:40.295064Z","iopub.execute_input":"2021-07-16T06:00:40.295462Z","iopub.status.idle":"2021-07-16T06:00:40.424241Z","shell.execute_reply.started":"2021-07-16T06:00:40.295425Z","shell.execute_reply":"2021-07-16T06:00:40.423238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create the label column \n\ndf_final.head()\n#len(df_final)","metadata":{"execution":{"iopub.status.busy":"2021-07-16T06:00:41.819083Z","iopub.execute_input":"2021-07-16T06:00:41.819421Z","iopub.status.idle":"2021-07-16T06:00:41.83801Z","shell.execute_reply.started":"2021-07-16T06:00:41.819393Z","shell.execute_reply":"2021-07-16T06:00:41.836895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_final","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Imports for libraries","metadata":{}},{"cell_type":"code","source":"import os\n\nimport torch\nimport torchvision\nfrom torchvision import datasets\nimport torchvision.transforms as transforms\n\nimport torchvision.models as models\nimport torch.optim as optim\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.autograd import Variable","metadata":{"execution":{"iopub.status.busy":"2021-07-16T06:00:44.982891Z","iopub.execute_input":"2021-07-16T06:00:44.983258Z","iopub.status.idle":"2021-07-16T06:00:44.989366Z","shell.execute_reply.started":"2021-07-16T06:00:44.983228Z","shell.execute_reply":"2021-07-16T06:00:44.988033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#define the class for the image dataset\nfrom torch.utils.data import Dataset, DataLoader\nfrom skimage import io\nfrom PIL import Image\n\nclass CustomImageDataset(Dataset):\n    \"\"\"CustomeImageDataset dataset.\"\"\"\n\n    def __init__(self, df, root_dir, transform=None):\n       \n        self.df = df\n        self.root_dir = root_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        #if torch.is_tensor(idx):\n        #    idx = idx.tolist()\n\n        img_name = os.path.join(self.root_dir,self.df.loc[self.df.index[idx], 'img_id'])\n        image = Image.open(img_name)\n        label = self.df.loc[self.df.index[idx], 'label']\n\n        if self.transform:\n            image = self.transform(image)\n\n        return (image, label)\n","metadata":{"execution":{"iopub.status.busy":"2021-07-16T06:00:48.08026Z","iopub.execute_input":"2021-07-16T06:00:48.080735Z","iopub.status.idle":"2021-07-16T06:00:48.090725Z","shell.execute_reply.started":"2021-07-16T06:00:48.080706Z","shell.execute_reply":"2021-07-16T06:00:48.08942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#using compose to make transformations on images \ntransform = {\n    \n    'train': transforms.Compose([transforms.ToTensor()\n                                  #transforms.Resize(256),\n                                  #transforms.RandomResizedCrop(224),\n                                  #transforms.RandomHorizontalFlip(),\n                                  #transforms.RandomRotation(10),\n                                  #normalization\n                                ]),\n    'valid': transforms.Compose([transforms.Resize(256),\n                                 transforms.CenterCrop(224),\n                                 transforms.ToTensor(),\n                                 #normalization\n                                ]),\n    'test': transforms.Compose([transforms.Resize(size=(224,224)),\n                                transforms.ToTensor(),\n                                #normalization\n                               ])\n                  }\n\n#load our image data \ndataset = CustomImageDataset(\n    df=df_final,\n    root_dir=\"../input/siimcovid19detectionjpg/archive (2)/archive/train/train-imgs\",\n    transform=transforms.ToTensor(),\n    #transform = transform[\"train\"]\n)\n\n#random spit into train and validation set \ntrain_set, validation_set = torch.utils.data.random_split(dataset, [round(.80*len(dataset)), round(.20*len(dataset))],generator=torch.Generator().manual_seed(42))\n\n#into the dataloader\ntrain_loader = DataLoader(dataset=train_set, batch_size=5, shuffle=True)\nval_loader = DataLoader(dataset=validation_set, batch_size=5, shuffle=True)\n","metadata":{"execution":{"iopub.status.busy":"2021-07-16T06:00:51.45537Z","iopub.execute_input":"2021-07-16T06:00:51.45584Z","iopub.status.idle":"2021-07-16T06:00:51.466609Z","shell.execute_reply.started":"2021-07-16T06:00:51.455776Z","shell.execute_reply":"2021-07-16T06:00:51.465142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# easy way to get some images ","metadata":{}},{"cell_type":"code","source":"# get some random training images\nimport numpy as np\nimport matplotlib.pyplot as plt\ndataiter = iter(train_loader)\nimages, labels = dataiter.next()\n\ndef imshow(img):\n    #img = img / 2 + 0.5  # unnormalize why ?\n    npimg = img.numpy()\n    plt.imshow(np.transpose(npimg, (1, 2, 0))) # why ?\n    plt.show() \n    \n# show images\nimshow(torchvision.utils.make_grid(images))  \n\n#conv = torch.nn.Conv2d(\n#\t\tin_channels=3, # RGB channels\n#\t\tout_channels=7, # Number of kernels\n#\t\tkernel_size=5, # Size of kernels, i. e. of size 5x5\n#\t\tstride=1,\n#\t\tpadding=2)\n\nconv1 = nn.Conv2d(1,6,5)\npool = nn.MaxPool2d(2,2)\nconv2 = nn.Conv2d(6, 16, 5)\nprint(images.shape)\n\nx = conv1(images)\nprint(x.shape)\nx = pool(x)\nprint(x.shape)\nx = conv2(x)\nprint(x.shape)\nx = pool(x)\nprint(x.shape)\n\n","metadata":{"execution":{"iopub.status.busy":"2021-07-16T06:00:55.18621Z","iopub.execute_input":"2021-07-16T06:00:55.186609Z","iopub.status.idle":"2021-07-16T06:00:55.530944Z","shell.execute_reply.started":"2021-07-16T06:00:55.186578Z","shell.execute_reply":"2021-07-16T06:00:55.529496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n#from matplotlib.pyplot import figure\nplt.rcParams[\"figure.figsize\"] = (10,10)\nfor batch_idx, (data, targets) in enumerate(train_loader):\n    print(targets)\n    print(data.shape)\n    #print(max(data[5][0][0]))\n    print(targets[0])\n    print(data[0].shape[1])\n    \"Simple – imshow expects images to be structured as \\\n    (rows, columns) for grayscale data and (rows, columns, channels) and possibly (rows, columns, channels, alpha) values for RGB(A) data.\\\n    You will thus have to reshape your grayscale visualization image into (256, 256) to make it work.\"\n    plt.imshow(data[0].reshape(data[0].shape[1],data[0].shape[1]),cmap='gray',interpolation='none')\n    break","metadata":{"execution":{"iopub.status.busy":"2021-07-16T05:26:53.611208Z","iopub.execute_input":"2021-07-16T05:26:53.611828Z","iopub.status.idle":"2021-07-16T05:26:53.906884Z","shell.execute_reply.started":"2021-07-16T05:26:53.611786Z","shell.execute_reply":"2021-07-16T05:26:53.905616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#class_ = (1, 2, 3, 4)\n#np.asarray(class_)\n#torch.from_numpy(np.asarray(class_))","metadata":{"execution":{"iopub.status.busy":"2021-07-16T05:18:16.513622Z","iopub.execute_input":"2021-07-16T05:18:16.514118Z","iopub.status.idle":"2021-07-16T05:18:16.532951Z","shell.execute_reply.started":"2021-07-16T05:18:16.514065Z","shell.execute_reply":"2021-07-16T05:18:16.531451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#build a CNN\n#imports are made in the above cell\nCUDA_LAUNCH_BLOCKING=1\n\n# Device configuration\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\n# Hyper-parameters \nnum_epochs = 5\nbatch_size = 4\nlearning_rate = 0.001\n\n#why normalize between [-1,1] ?\n\n#classes = ('typical', 'no_covid', 'atypical', 'intermediate')\n\nclass ConvNet(nn.Module):\n    def __init__(self):\n        super(ConvNet, self).__init__()\n        self.conv1 = nn.Conv2d(1, 6, 5)\n        self.pool = nn.MaxPool2d(2, 2)\n        self.conv2 = nn.Conv2d(6, 16, 5)\n        self.fc1 = nn.Linear(16 * 61 * 61, 120)\n        self.fc2 = nn.Linear(120, 84)\n        self.fc3 = nn.Linear(84, 4 )\n\n    def forward(self, x):\n        # -> n, 3, 32, 32\n        x = self.pool(F.relu(self.conv1(x)))  \n        x = self.pool(F.relu(self.conv2(x))) \n        x = x.view(-1, 16 * 61 * 61)            \n        x = F.relu(self.fc1(x))               \n        x = F.relu(self.fc2(x))              \n        x = self.fc3(x)                       \n        return x\n\n\nmodel = ConvNet()\ncriterion = nn.CrossEntropyLoss()\noptimizer = torch.optim.SGD(model.parameters(), lr=learning_rate)\n\nn_total_steps = len(train_loader)\nfor epoch in range(num_epochs):\n    for i, (images, labels) in enumerate(train_loader):\n        #print(images.shape, labels)\n        # origin shape: [5, 1, 256, 256] = 5,1,256*256\n        # input_layer: 1 input channels, 4 output channels, 5 kernel size\n        images = images\n        labels = labels\n        #print(\"hi\")\n        # Forward pass\n        outputs = model(images)\n        loss = criterion(outputs, labels)\n        #print(\"hiello\")\n\n        # Backward and optimize\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n        #print(\"yo\")\n        if (i+1) % 10 == 0:\n            print (f'Epoch [{epoch+1}/{num_epochs}], Step [{i+1}/{n_total_steps}], Loss: {loss.item():.4f}')\nprint('Finished Training')\n","metadata":{"execution":{"iopub.status.busy":"2021-07-16T06:37:36.692665Z","iopub.execute_input":"2021-07-16T06:37:36.693118Z","iopub.status.idle":"2021-07-16T06:48:44.731915Z","shell.execute_reply.started":"2021-07-16T06:37:36.693089Z","shell.execute_reply":"2021-07-16T06:48:44.729788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test it out\ncorrect = 0\ntotal = 0\n# since we're not training, we don't need to calculate the gradients for our outputs\nwith torch.no_grad():\n    for data in val_loader:\n        images, labels = data\n        # calculate outputs by running images through the network\n        outputs = model(images)\n        # the class with the highest energy is what we choose as prediction\n        _, predicted = torch.max(outputs.data, 1)\n        total += labels.size(0)\n        correct += (predicted == labels).sum().item()\n\nprint('Accuracy of the network on the test images: %d %%' % (\n    100 * correct / total))","metadata":{"execution":{"iopub.status.busy":"2021-07-16T06:56:04.481424Z","iopub.execute_input":"2021-07-16T06:56:04.481888Z","iopub.status.idle":"2021-07-16T06:56:15.96977Z","shell.execute_reply.started":"2021-07-16T06:56:04.481843Z","shell.execute_reply":"2021-07-16T06:56:15.96776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}