{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"https://www.kaggle.com/givkashi/transfer-learning-with-vgg19 # This uses TensorFlow Keras, but shows preprocessing, augmentation, dataloading, and model\n\nhttps://www.kaggle.com/pvtien96/siim-cov19-efnb7-yolov5-infer # overly complicated (maybe?) but potentially some good steps\n\nhttps://www.kaggle.com/allunia/pulmonary-dicom-preprocessing # great source for DICOM, but still in process\n\nhttps://www.kaggle.com/andradaolteanu/siim-covid-19-box-detect-dcm-metadata # the Extract Metadata from .dcm is pretty slick\n\nhttps://www.kaggle.com/ayuraj/train-covid-19-detection-using-yolov5 # great section on dataloading and bounding boxes","metadata":{}},{"cell_type":"markdown","source":"https://www.kaggle.com/piantic/train-siim-covid-19-detection-fasterrcnn # maybe a baseline for using pytorch for dataloading/model?\n\nhttps://www.kaggle.com/southsakura/covid19 # great notebook, similar to our projected structure (in Chinese and my chinese is extremely nonexistent)\n\nhttps://www.kaggle.com/ammarnassanalhajali/siim-fisabio-rsna-covid-19-detection-eda # super cool visualization based on labels","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"scrolled":true,"execution":{"iopub.status.busy":"2021-07-23T06:14:14.403422Z","iopub.execute_input":"2021-07-23T06:14:14.403756Z","iopub.status.idle":"2021-07-23T06:14:55.479061Z","shell.execute_reply.started":"2021-07-23T06:14:14.403725Z","shell.execute_reply":"2021-07-23T06:14:55.478239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Findings:\n\n1.There are duplicates: same StudyInstanceID can appear more than 1 <br>\n2.There are 280 duplicated entries <br>\n3.For the duplicated entries, the studyinstanceid remains same but the SOP id changes <br>\n4. Study Instance ID > Series Instance ID > SOP ID\n5.In particular, you’ll categorize the radiographs as negative for pneumonia or typical, indeterminate, or atypical for COVID-19. You and your model will work with <br>imaging data and annotations from a group of radiologists.<br>\n\n> **What to do next  ????** \n\n1. conditional filtering jpegs for unique values\n\n\n","metadata":{}},{"cell_type":"markdown","source":"# Reading the study file","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n## taking a look at the study level data \ndf_TSL = pd.read_csv('../input/siim-covid19-detection/train_study_level.csv')\n\n#dividing the column name into two\ndf_TSL[[\"StudyInstanceUID\",\"id\"]] = df_TSL[\"id\"].str.split(\"_\", expand = True)\ndf_TSL.drop([\"id\"],axis = 1, inplace = True)\ndf_TSL.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:14:55.481977Z","iopub.execute_input":"2021-07-23T06:14:55.482251Z","iopub.status.idle":"2021-07-23T06:14:55.547191Z","shell.execute_reply.started":"2021-07-23T06:14:55.482224Z","shell.execute_reply":"2021-07-23T06:14:55.546235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reading the image file","metadata":{}},{"cell_type":"code","source":"df_TIL = pd.read_csv('../input/siim-covid19-detection/train_image_level.csv')\n\ndf_TIL[[\"SOP_id\", \"id\"]] = df_TIL.id.str.split(\"_\", expand = True,)\ndf_TIL.drop([\"id\"],inplace =True, axis =1)\ndf_TIL.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:14:55.549154Z","iopub.execute_input":"2021-07-23T06:14:55.549526Z","iopub.status.idle":"2021-07-23T06:14:55.619337Z","shell.execute_reply.started":"2021-07-23T06:14:55.549485Z","shell.execute_reply":"2021-07-23T06:14:55.618342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Joining the two","metadata":{}},{"cell_type":"code","source":"#join the two tables\n#df_TIL.join(df_TSL, on=\"StudyInstanceUID\")\ndf_TIL_TSL = pd.merge(\n    df_TIL, \n    df_TSL, \n    left_on=\"StudyInstanceUID\", \n    right_on=\"StudyInstanceUID\", \n    how=\"left\", \n    sort=False\n)\n\ndf_TIL_TSL.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:14:55.620935Z","iopub.execute_input":"2021-07-23T06:14:55.621292Z","iopub.status.idle":"2021-07-23T06:14:55.649017Z","shell.execute_reply.started":"2021-07-23T06:14:55.621254Z","shell.execute_reply":"2021-07-23T06:14:55.648192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# conditional filtering jpegs for unique values - Step 1","metadata":{}},{"cell_type":"code","source":"print(len(df_TIL_TSL))\ndf_SOP_noduplicate = df_TIL_TSL[df_TIL_TSL['StudyInstanceUID'].map((df_TIL_TSL['StudyInstanceUID'].value_counts()) > 1) & (df_TIL_TSL['Negative for Pneumonia'] != 1)].dropna()\nl_duplicates = list(df_SOP_noduplicate.StudyInstanceUID.value_counts().keys())\ndf = df_TIL_TSL[~df_TIL_TSL['StudyInstanceUID'].isin(l_duplicates)]\ndf = df.append(df_SOP_noduplicate)\n\nd=pd.DataFrame()\nfor name, data in df.groupby([\"StudyInstanceUID\"]):\n    if len(data)>1:\n        #print(len(data))\n        df = df[df.StudyInstanceUID != name]\n        d = d.append(data.sample())\n\n\ndf_final = df.append(d)\nprint(len(df_final))\n","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:14:55.650385Z","iopub.execute_input":"2021-07-23T06:14:55.650851Z","iopub.status.idle":"2021-07-23T06:14:56.173295Z","shell.execute_reply.started":"2021-07-23T06:14:55.65074Z","shell.execute_reply":"2021-07-23T06:14:56.171129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_final.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-23T03:01:12.911625Z","iopub.execute_input":"2021-07-23T03:01:12.91213Z","iopub.status.idle":"2021-07-23T03:01:12.925859Z","shell.execute_reply.started":"2021-07-23T03:01:12.912097Z","shell.execute_reply":"2021-07-23T03:01:12.924739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Plt the images","metadata":{}},{"cell_type":"code","source":"#Let's bring the pydicom to view the images \n#!pip install pydicom\n\nimport pydicom\n\ndef dicom2array(path, voi_lut=True, fix_monochrome=True):\n    dicom = pydicom.read_file(path)\n    # VOI LUT (if available by DICOM device) is used to\n    # transform raw DICOM data to \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n        \ndef plot_img(img, size=(7, 7), is_rgb=True, title=\"\", cmap='gray'):\n    plt.figure(figsize=size)\n    plt.imshow(img, cmap=cmap)\n    plt.suptitle(title)\n    plt.show()\n\n\ndef plot_imgs(imgs, cols=4, size=7, is_rgb=True, title=\"\", cmap='gray', img_size=(500,500)):\n    rows = len(imgs)//cols + 1\n    fig = plt.figure(figsize=(cols*size, rows*size))\n    for i, img in enumerate(imgs):\n        if img_size is not None:\n            img = cv2.resize(img, img_size)\n        fig.add_subplot(rows, cols, i+1)\n        plt.imshow(img, cmap=cmap)\n    plt.suptitle(title)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:14:56.177446Z","iopub.execute_input":"2021-07-23T06:14:56.179755Z","iopub.status.idle":"2021-07-23T06:14:56.407471Z","shell.execute_reply.started":"2021-07-23T06:14:56.17971Z","shell.execute_reply":"2021-07-23T06:14:56.406505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sample_dcm = pydicom.read_file('../input/siim-covid19-detection/train/00086460a852/9e8302230c91/65761e66de9f.dcm')\n#sample_dcm","metadata":{"execution":{"iopub.status.busy":"2021-07-17T04:45:45.347817Z","iopub.execute_input":"2021-07-17T04:45:45.34832Z","iopub.status.idle":"2021-07-17T04:45:45.352116Z","shell.execute_reply.started":"2021-07-17T04:45:45.348283Z","shell.execute_reply":"2021-07-17T04:45:45.351225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"import os\nlst_study =[]\nlst_series = []\nlst_dcm =[]\ntrain_img_path = pd.DataFrame(columns = [\"StudyInstanceUID\" ,\"SeriesInstanceUID\" , \"SOP_dcm\"])\nfor filename in os.listdir(\"../input/siim-covid19-detection/train\"):\n    #lst_study.append(filename)\n    for file in os.listdir(\"../input/siim-covid19-detection/train/\"+filename):\n        #lst_series.append(file)\n        for dcm in os.listdir(\"../input/siim-covid19-detection/train/\"+filename+\"/\"+file):\n            #lst_dcm.append(dcm)\n            train_img_path.loc[len(train_img_path)] = [filename ,file, dcm]\n            \ntrain_img_path['path_to_dcm'] = train_img_path[\"StudyInstanceUID\"] + \"/\"+ train_img_path[\"SeriesInstanceUID\"] + \"/\" +train_img_path[\"SOP_dcm\"] \n\n            \n    ","metadata":{"execution":{"iopub.status.busy":"2021-07-23T02:57:39.893315Z","iopub.execute_input":"2021-07-23T02:57:39.893673Z","iopub.status.idle":"2021-07-23T02:58:04.986893Z","shell.execute_reply.started":"2021-07-23T02:57:39.893643Z","shell.execute_reply":"2021-07-23T02:58:04.985927Z"}}},{"cell_type":"markdown","source":"train_img_path[[\"SOP_id\",\"dcm\"]] = train_img_path[\"SOP_dcm\"].str.split(\".\",expand = True)\ntrain_img_path.drop([\"dcm\"],inplace =True, axis =1)\ntrain_img_path.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-23T02:58:04.988677Z","iopub.execute_input":"2021-07-23T02:58:04.989068Z","iopub.status.idle":"2021-07-23T02:58:05.022434Z","shell.execute_reply.started":"2021-07-23T02:58:04.989025Z","shell.execute_reply":"2021-07-23T02:58:05.021404Z"}}},{"cell_type":"markdown","source":"#visualize the images for a particular study instance id which has duplicates \nimport matplotlib.pyplot as plt\n\nStudyInstanceUID =\"a7335b2f9815\"\ndf = train_img_path[train_img_path.StudyInstanceUID == StudyInstanceUID]\n\n\nfor i in df.path_to_dcm:\n        \n    sample_dcm = pydicom.read_file('../input/siim-covid19-detection/train/' + i)\n    \n    plot_img(sample_dcm.pixel_array)\n   ","metadata":{"execution":{"iopub.status.busy":"2021-07-17T04:46:06.102699Z","iopub.execute_input":"2021-07-17T04:46:06.103153Z","iopub.status.idle":"2021-07-17T04:46:15.931544Z","shell.execute_reply.started":"2021-07-17T04:46:06.103118Z","shell.execute_reply":"2021-07-17T04:46:15.930762Z"}}},{"cell_type":"markdown","source":"#join the image path with the dataset \ndf_final = pd.merge(\n    df_TIL_TSL, train_img_path[[\"SOP_id\" , \"path_to_dcm\",\"SeriesInstanceUID\"]], \n    left_on=\"SOP_id\", \n    right_on=\"SOP_id\", \n    how=\"left\", sort=False\n)\ndf_final.head()\n","metadata":{"execution":{"iopub.status.busy":"2021-07-23T02:58:50.128179Z","iopub.execute_input":"2021-07-23T02:58:50.128552Z","iopub.status.idle":"2021-07-23T02:58:50.158152Z","shell.execute_reply.started":"2021-07-23T02:58:50.128519Z","shell.execute_reply":"2021-07-23T02:58:50.157017Z"}}},{"cell_type":"markdown","source":"## Object Detection stuff \n### and things \n#### and jazz \n##### and yea!","metadata":{}},{"cell_type":"code","source":"pd.options.display.max_colwidth=250\n\nbbox_sample = df_final.loc[df_final.StudyInstanceUID==sample_dcm.StudyInstanceUID]\n\nbbox_sample_boxes = df_final.loc[df_final.StudyInstanceUID=='f57e0c92aa87']","metadata":{"execution":{"iopub.status.busy":"2021-07-17T04:46:15.992418Z","iopub.execute_input":"2021-07-17T04:46:15.99275Z","iopub.status.idle":"2021-07-17T04:46:16.001731Z","shell.execute_reply.started":"2021-07-17T04:46:15.992717Z","shell.execute_reply":"2021-07-17T04:46:16.000606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bbox_sample_boxes","metadata":{"execution":{"iopub.status.busy":"2021-07-17T04:46:16.003341Z","iopub.execute_input":"2021-07-17T04:46:16.003714Z","iopub.status.idle":"2021-07-17T04:46:16.018463Z","shell.execute_reply.started":"2021-07-17T04:46:16.003679Z","shell.execute_reply":"2021-07-17T04:46:16.017363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_dcm","metadata":{"execution":{"iopub.status.busy":"2021-07-17T04:46:16.024439Z","iopub.execute_input":"2021-07-17T04:46:16.024683Z","iopub.status.idle":"2021-07-17T04:46:16.033481Z","shell.execute_reply.started":"2021-07-17T04:46:16.02466Z","shell.execute_reply":"2021-07-17T04:46:16.032509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bounds = str(bbox_sample_boxes.iloc[:,0:1])\nsample_dcm_bounds = bbox_sample_boxes.iloc[:,:]\nsample_dcm_bounds_image = pydicom.read_file('../input/siim-covid19-detection/train/f57e0c92aa87/c748d4ba8a10/2881f01cf3f1.dcm')\ntype(bounds)","metadata":{"execution":{"iopub.status.busy":"2021-07-17T04:46:16.037052Z","iopub.execute_input":"2021-07-17T04:46:16.037336Z","iopub.status.idle":"2021-07-17T04:46:16.087376Z","shell.execute_reply.started":"2021-07-17T04:46:16.037306Z","shell.execute_reply":"2021-07-17T04:46:16.086634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# stolen from SO, i have no idea how it works... whooooooooooooooooooooooooooooops\nimport re\nimport ast\nx = ast.literal_eval(re.search('({.+})', bounds).group(0))\ndf_x = pd.DataFrame(x)\n# x[0] # shows the array","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2021-07-17T04:46:16.089927Z","iopub.execute_input":"2021-07-17T04:46:16.090189Z","iopub.status.idle":"2021-07-17T04:46:16.097481Z","shell.execute_reply.started":"2021-07-17T04:46:16.090164Z","shell.execute_reply":"2021-07-17T04:46:16.096647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install regions","metadata":{"execution":{"iopub.status.busy":"2021-07-17T04:46:16.100441Z","iopub.execute_input":"2021-07-17T04:46:16.10073Z","iopub.status.idle":"2021-07-17T04:46:41.570133Z","shell.execute_reply.started":"2021-07-17T04:46:16.1007Z","shell.execute_reply":"2021-07-17T04:46:41.569041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#this is an empty plot, BUT it shows that the dictionary values can be extracted\n# plt.plot(x[0].get('x'),x[0].get('y'),x[0].get('width'),x[0].get('height'),'k-',lw=2)\n\n#import cv2\n#img = cv2.imread(sample_dcm)\n#cv2.rectangle(im,(x[0].get('x'),x[0].get('y')),(x[0].get('width'),x[0].get('height')),(0,255,0),2)\n\n# https://www.kaggle.com/kiwifairy/visualize-x-ray-image-with-bounding-boxes\n\nimport matplotlib.patches as patches\nfrom regions import BoundingBox\n\ndef plot_img_bbox(img, boxes_df):\n    # plot the image and bboxes\n    # Bounding boxes are defined as follows: x-min y-min width height\n    fig, a = plt.subplots(1,1)\n    fig.set_size_inches(10,10)\n    a.imshow(img, cmap = 'gray')\n    for index, row in boxes_df.iterrows():\n        x, y, width, height  = row['x'], row['y'], row['width'], row['height']\n        \"\"\"rect = patches.Rectangle((x, y),\n                                 x+width, y+height,\n                                 linewidth = 2,\n                                 edgecolor = 'w',\n                                 facecolor = 'none')\"\"\"\n        \n        bbox = BoundingBox(ixmin=int(x), ixmax=int(x+width), iymin=int(y), iymax=int(y+height))\n\n        # Draw the bounding box on top of the image\n        #a.add_patch(rect)\n        a.add_patch(bbox.as_artist(facecolor='none', edgecolor='white', lw=2.))\n    plt.show()\n    \n\nplot_img_bbox(sample_dcm_bounds_image.pixel_array, df_x)","metadata":{"execution":{"iopub.status.busy":"2021-07-17T04:46:41.57183Z","iopub.execute_input":"2021-07-17T04:46:41.572232Z","iopub.status.idle":"2021-07-17T04:46:44.956527Z","shell.execute_reply.started":"2021-07-17T04:46:41.572189Z","shell.execute_reply":"2021-07-17T04:46:44.952145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dataloader for DICOM","metadata":{}},{"cell_type":"code","source":"import torch","metadata":{"execution":{"iopub.status.busy":"2021-07-17T04:46:44.967117Z","iopub.execute_input":"2021-07-17T04:46:44.967604Z","iopub.status.idle":"2021-07-17T04:46:44.978544Z","shell.execute_reply.started":"2021-07-17T04:46:44.967568Z","shell.execute_reply":"2021-07-17T04:46:44.977389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_train_file_path(image_id):\n    dicom_file =  f\"../input/siim-covid19-detection/train/{image_id}\"\n    \n    dmc_image = pydicom.read_file(dicom_file)\n    \n    return dmc_image","metadata":{"execution":{"iopub.status.busy":"2021-07-17T04:46:44.99065Z","iopub.execute_input":"2021-07-17T04:46:44.991338Z","iopub.status.idle":"2021-07-17T04:46:44.999469Z","shell.execute_reply.started":"2021-07-17T04:46:44.991293Z","shell.execute_reply":"2021-07-17T04:46:44.998574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = get_train_file_path(df_final.path_to_dcm[0])","metadata":{"execution":{"iopub.status.busy":"2021-07-17T04:46:45.000944Z","iopub.execute_input":"2021-07-17T04:46:45.001533Z","iopub.status.idle":"2021-07-17T04:46:46.091369Z","shell.execute_reply.started":"2021-07-17T04:46:45.00148Z","shell.execute_reply":"2021-07-17T04:46:46.090506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_img(test.pixel_array)","metadata":{"execution":{"iopub.status.busy":"2021-07-17T04:46:46.092569Z","iopub.execute_input":"2021-07-17T04:46:46.092909Z","iopub.status.idle":"2021-07-17T04:46:47.514456Z","shell.execute_reply.started":"2021-07-17T04:46:46.092872Z","shell.execute_reply":"2021-07-17T04:46:47.513467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"https://www.kaggle.com/xhlulu/siim-covid19-resized-to-256px-jpg # maybe for use of a single ImageFolder Directory","metadata":{}},{"cell_type":"markdown","source":"# DataLoader and Dataset","metadata":{}},{"cell_type":"markdown","source":"This cell creates the normalization for RGB, the number of workers, batch size, and the transform separated for train/valid/test (i.e., `transform['train']`)","metadata":{}},{"cell_type":"code","source":"import os\n\nimport torch\nimport torchvision\nfrom torchvision import datasets\nimport torchvision.transforms as transforms\n\nimport torchvision.models as models\nimport torch.optim as optim\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.autograd import Variable\n","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:14:56.410904Z","iopub.execute_input":"2021-07-23T06:14:56.411238Z","iopub.status.idle":"2021-07-23T06:14:57.721879Z","shell.execute_reply.started":"2021-07-23T06:14:56.411209Z","shell.execute_reply":"2021-07-23T06:14:57.721035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_final.shape # check if 6054, this means that it is only unique/no duplicates","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:14:57.725907Z","iopub.execute_input":"2021-07-23T06:14:57.72622Z","iopub.status.idle":"2021-07-23T06:14:57.734487Z","shell.execute_reply.started":"2021-07-23T06:14:57.726191Z","shell.execute_reply":"2021-07-23T06:14:57.733635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_final.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-23T04:57:32.208721Z","iopub.execute_input":"2021-07-23T04:57:32.209252Z","iopub.status.idle":"2021-07-23T04:57:32.225992Z","shell.execute_reply.started":"2021-07-23T04:57:32.209218Z","shell.execute_reply":"2021-07-23T04:57:32.224741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we need the ids of the imgs to properly join to the parent directory for dataloading\ndf_final['img_id'] = df_final['SOP_id'].astype('str') +'.jpg'","metadata":{"execution":{"iopub.status.busy":"2021-07-23T03:01:27.101852Z","iopub.execute_input":"2021-07-23T03:01:27.102192Z","iopub.status.idle":"2021-07-23T03:01:27.109585Z","shell.execute_reply.started":"2021-07-23T03:01:27.102163Z","shell.execute_reply":"2021-07-23T03:01:27.108522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dip added \nEjm copied/pasted Dip's section :)","metadata":{}},{"cell_type":"code","source":"def label_race (row):\n    if row['Negative for Pneumonia'] == 1 :\n        return 0\n    if row['Typical Appearance'] == 1 :\n        return 1\n    if row['Indeterminate Appearance'] == 1 :\n        return 2\n    if row['Atypical Appearance'] == 1 :\n        return 3\n   \n    return 'Other'\n\ndf_final['label'] = df_final.apply (lambda row: label_race(row), axis=1)\ndf_final.drop([\"Negative for Pneumonia\",\"Typical Appearance\",\"Indeterminate Appearance\",\"Atypical Appearance\"], axis =1, inplace = True)\ndf_final['img_id'] = df_final['SOP_id'].astype('str') +'.jpg'","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:15:27.459494Z","iopub.execute_input":"2021-07-23T06:15:27.459826Z","iopub.status.idle":"2021-07-23T06:15:27.660982Z","shell.execute_reply.started":"2021-07-23T06:15:27.459795Z","shell.execute_reply":"2021-07-23T06:15:27.660154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_final.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:15:29.533936Z","iopub.execute_input":"2021-07-23T06:15:29.534304Z","iopub.status.idle":"2021-07-23T06:15:29.545068Z","shell.execute_reply.started":"2021-07-23T06:15:29.534275Z","shell.execute_reply":"2021-07-23T06:15:29.544229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classes = ['Negative', 'Typical', 'Indeterminate', 'Atypical']","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:15:29.968318Z","iopub.execute_input":"2021-07-23T06:15:29.968655Z","iopub.status.idle":"2021-07-23T06:15:29.972765Z","shell.execute_reply.started":"2021-07-23T06:15:29.968625Z","shell.execute_reply":"2021-07-23T06:15:29.971351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"https://towardsdatascience.com/how-to-load-a-custom-image-dataset-on-pytorch-bf10b2c529e0https://towardsdatascience.com/how-to-load-a-custom-image-dataset-on-pytorch-bf10b2c529e0\n\nhttps://www.kaggle.com/leifuer/intro-to-pytorch-loading-image-data\n\nhttps://pytorch.org/vision/stable/datasets.html\n\nhttps://pytorch.org/tutorials/beginner/basics/data_tutorial.html","metadata":{}},{"cell_type":"code","source":"# this is fabricated from the above hyperlinks, i don't claim to fully understand it\n\nfrom torch.utils.data import Dataset, DataLoader\n\nfrom PIL import Image\n\nclass CustomImageDataset(Dataset):\n    \"\"\"CustomeImageDataset dataset.\"\"\"\n\n    def __init__(self, df, root_dir, transform=None):\n        \"\"\"\n        Args:\n            img_filename (string): Path to the img_filename file .\n            root_dir (string): Directory with all the images.\n            transform (callable, optional): Optional transform to be applied\n                on a sample.\n        \"\"\"\n        self.df = df\n        self.root_dir = root_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        # if torch.is_tensor(idx):\n        #     idx = idx.tolist()\n\n        img_name = os.path.join(self.root_dir,\n                                self.df.loc[self.df.index[idx], 'img_id'])\n        image = Image.open(img_name)\n        image = image.convert(\"RGB\") # need to track down why the greyscale represents as RGB\n        label = self.df.loc[self.df.index[idx], 'label']\n        \n        label = torch.tensor(label)\n        \n        if self.transform:\n            image = self.transform(image)\n\n        return (image, label)","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:15:32.440555Z","iopub.execute_input":"2021-07-23T06:15:32.44088Z","iopub.status.idle":"2021-07-23T06:15:32.450026Z","shell.execute_reply.started":"2021-07-23T06:15:32.440842Z","shell.execute_reply":"2021-07-23T06:15:32.449064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# RGB\n# normalization = transforms.Normalize(mean=[0.485, 0.456, 0.406],\n#                                              std=[0.229, 0.224, 0.225])\n\n# Greyscale (broadcast shape of 1, 224,224)\n\n# normalization = transforms.Normalize(mean=[0.5],\n#                                              std=[0.225])\n\nnum_workers = 0\nbatch_size = 19\n\ntransform = {\n    \n    'train': transforms.Compose([transforms.Resize(256),\n                                  transforms.CenterCrop(224),\n                                  transforms.RandomHorizontalFlip(),\n                                  transforms.RandomRotation(2),\n                                  transforms.ToTensor(),\n                                  #normalization\n                                ]),\n    'valid': transforms.Compose([transforms.Resize(256),\n                                 transforms.CenterCrop(224),\n                                 transforms.ToTensor(),\n                                 #normalization\n                                ]),\n    'test': transforms.Compose([transforms.Resize(size=(224,224)),\n                                transforms.ToTensor(),\n                                #normalization\n                               ])\n                  }\n  ","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:18:07.318811Z","iopub.execute_input":"2021-07-23T06:18:07.319174Z","iopub.status.idle":"2021-07-23T06:18:07.327016Z","shell.execute_reply.started":"2021-07-23T06:18:07.319141Z","shell.execute_reply":"2021-07-23T06:18:07.325903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"-# need home directory for CustomImageDataset\n\n-# train\ntrain_dir = \"../input/siimcovid19detectionjpg/archive (2)/archive/train/train-imgs\"\n\n-# test\ntest_dir = \"../input/siimcovid19detectionjpg/archive (2)/archive/train/test-imgs\"\n\n-# validate # this requires a subsetting/sampling from train directory\n-# validate_dir = \"../input/siimcovid19detectionjpg/archive (2)/archive/train/train-imgs\"\n\n\nimages=[]\nimages_ids = []\n\nfor file in os.listdir(train_dir):\n    if file.endswith(\".jpg\"):\n        images.append(file)\n        \nfor file in images:\n    images_ids.append(os.path.splitext(file)[0])\n       \nfor img in images:\n\n    img_id = os.path.splitext(img)[0]\n    if img_id in images_ids:\n        \n        train_data = CustomImageDataset(\n            df_final, \n            train_dir,\n            transform=transform['train']\n        )","metadata":{"execution":{"iopub.status.busy":"2021-07-23T04:58:00.696503Z","iopub.execute_input":"2021-07-23T04:58:00.696928Z","iopub.status.idle":"2021-07-23T04:58:01.31163Z","shell.execute_reply.started":"2021-07-23T04:58:00.696893Z","shell.execute_reply":"2021-07-23T04:58:01.310311Z"}}},{"cell_type":"code","source":"#load our image data \ndataset = CustomImageDataset(\n    df=df_final,\n    root_dir=\"../input/siimcovid19detectionjpg/archive (2)/archive/train/train-imgs\",\n    transform=transforms.ToTensor(),\n    #transform = transform[\"train\"]\n)\n\n#random spit into train and validation set \ntrain_set, validation_set = torch.utils.data.random_split(dataset, [round(.80*len(dataset)), round(.20*len(dataset))],generator=torch.Generator().manual_seed(42))\n\n#into the dataloader\ntrain_loader = DataLoader(dataset=train_set, batch_size=19, shuffle=True)\nval_loader = DataLoader(dataset=validation_set, batch_size=19, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:20:01.338511Z","iopub.execute_input":"2021-07-23T06:20:01.338848Z","iopub.status.idle":"2021-07-23T06:20:01.34631Z","shell.execute_reply.started":"2021-07-23T06:20:01.338817Z","shell.execute_reply":"2021-07-23T06:20:01.345043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for img, label in train_loader:\n    print(type(img))\n    print(type(label))\n    break","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:15:39.850449Z","iopub.execute_input":"2021-07-23T06:15:39.85085Z","iopub.status.idle":"2021-07-23T06:15:39.949307Z","shell.execute_reply.started":"2021-07-23T06:15:39.850814Z","shell.execute_reply":"2021-07-23T06:15:39.948331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_final.iloc[0]","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:15:46.313975Z","iopub.execute_input":"2021-07-23T06:15:46.31431Z","iopub.status.idle":"2021-07-23T06:15:46.321311Z","shell.execute_reply.started":"2021-07-23T06:15:46.314281Z","shell.execute_reply":"2021-07-23T06:15:46.320253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\ntemp_img, temp_label = train_data[0]\nplt.imshow(temp_img.numpy().transpose((1,2,0)))\nplt.title(temp_label)","metadata":{"execution":{"iopub.status.busy":"2021-07-17T04:46:48.529102Z","iopub.execute_input":"2021-07-17T04:46:48.529373Z","iopub.status.idle":"2021-07-17T04:46:48.68674Z","shell.execute_reply.started":"2021-07-17T04:46:48.529348Z","shell.execute_reply":"2021-07-17T04:46:48.685778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"train_data = datasets.ImageFolder(\"../input/siimcovid19detectionjpg/archive (2)/archive/train\", \n                                  transform=transform['train'])\n\n#valid_data = datasets.ImageFolder(\"../input/siimcovid19detectionjpg\", \n                                              #transform=transform['valid'])\n\ntest_data = datasets.ImageFolder(\"../input/siimcovid19detectionjpg/archive (2)/archive/test\", \n                                             transform=transform['test'])\n    ","metadata":{"execution":{"iopub.status.busy":"2021-07-17T04:46:48.688246Z","iopub.execute_input":"2021-07-17T04:46:48.688585Z","iopub.status.idle":"2021-07-17T04:46:48.739395Z","shell.execute_reply.started":"2021-07-17T04:46:48.688547Z","shell.execute_reply":"2021-07-17T04:46:48.738522Z"}}},{"cell_type":"code","source":"# THIS IS JUST A TEST CELL ;)\nfrom torch.utils.data import random_split\n\ntester, trainer = random_split(range(20), [2, 18], generator=torch.Generator().manual_seed(42))\n\nprint(tester.indices)\n\nprint(trainer.indices)\n\ntype(tester)\n# creates Subset - torch.utils.data.dataset.Subset\n\ndata_size = len(train_data)\n\nvalid_size = round(data_size *.2)\n\ntrain_size = round(data_size*.8)\n\ntrain_data_1, valid_data_1 = random_split(train_data, [train_size,valid_size], generator=torch.Generator().manual_seed(42))","metadata":{"execution":{"iopub.status.busy":"2021-07-23T05:01:23.623788Z","iopub.execute_input":"2021-07-23T05:01:23.624198Z","iopub.status.idle":"2021-07-23T05:01:23.631137Z","shell.execute_reply.started":"2021-07-23T05:01:23.624162Z","shell.execute_reply":"2021-07-23T05:01:23.629483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\ntemp_img, temp_label = train_data_1[0]\nplt.imshow(temp_img.numpy().transpose((1,2,0)))\nplt.title(temp_label)","metadata":{"execution":{"iopub.status.busy":"2021-07-23T05:01:25.765989Z","iopub.execute_input":"2021-07-23T05:01:25.766389Z","iopub.status.idle":"2021-07-23T05:01:25.98751Z","shell.execute_reply.started":"2021-07-23T05:01:25.766356Z","shell.execute_reply":"2021-07-23T05:01:25.986156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp_img, temp_label = valid_data_1[0]\nplt.imshow(temp_img.numpy().transpose((1,2,0)))\nplt.title(temp_label)","metadata":{"execution":{"iopub.status.busy":"2021-07-23T05:01:28.97179Z","iopub.execute_input":"2021-07-23T05:01:28.9722Z","iopub.status.idle":"2021-07-23T05:01:29.154201Z","shell.execute_reply.started":"2021-07-23T05:01:28.972167Z","shell.execute_reply":"2021-07-23T05:01:29.153442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_loader = torch.utils.data.DataLoader(train_data_1, \n                                           batch_size=batch_size,\n                                           num_workers=num_workers, \n                                           shuffle=True)\nvalid_loader = torch.utils.data.DataLoader(valid_data_1, \n                                           batch_size=batch_size,\n                                           num_workers=num_workers,\n                                           shuffle=True)\n#test_loader = torch.utils.data.DataLoader(test_data, \n#                                          batch_size=batch_size,\n#                                          num_workers=num_workers,\n#                                          shuffle=True)\n#\ndata_loaders = {'train': train_loader, \n                   'valid': valid_loader, \n                  # 'test': test_loader\n               }","metadata":{"execution":{"iopub.status.busy":"2021-07-23T05:02:12.188197Z","iopub.execute_input":"2021-07-23T05:02:12.188764Z","iopub.status.idle":"2021-07-23T05:02:12.19458Z","shell.execute_reply.started":"2021-07-23T05:02:12.188705Z","shell.execute_reply":"2021-07-23T05:02:12.19361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'train dataset: {len(train_data_1)}')\nprint(f'valid dataset: {len(valid_data_1)}')\n# print(f'test dataset: {len(test_data)}')\n# print(f'classes: {len(train_loader.classes)}')","metadata":{"execution":{"iopub.status.busy":"2021-07-23T05:02:54.096491Z","iopub.execute_input":"2021-07-23T05:02:54.097094Z","iopub.status.idle":"2021-07-23T05:02:54.104069Z","shell.execute_reply.started":"2021-07-23T05:02:54.09704Z","shell.execute_reply":"2021-07-23T05:02:54.102574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"next(iter(train_loader))[-1]","metadata":{"execution":{"iopub.status.busy":"2021-07-23T05:44:25.538635Z","iopub.execute_input":"2021-07-23T05:44:25.539083Z","iopub.status.idle":"2021-07-23T05:44:25.568242Z","shell.execute_reply.started":"2021-07-23T05:44:25.539049Z","shell.execute_reply":"2021-07-23T05:44:25.567369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\narr =train_data[0][0]\ndef plot_imgs(img, cols=4, size=7, is_rgb=True, title=\"\", cmap='gray', img_size=(500,500)):\n    img = cv2.resize(img, img_size)\n    plt.imshow(img, cmap=cmap)\n    plt.show()\nplt.imshow(np.transpose(train_data[0][0].numpy(), (1, 2, 0)))","metadata":{"execution":{"iopub.status.busy":"2021-07-23T05:03:18.599824Z","iopub.execute_input":"2021-07-23T05:03:18.600594Z","iopub.status.idle":"2021-07-23T05:03:18.80789Z","shell.execute_reply.started":"2021-07-23T05:03:18.600545Z","shell.execute_reply":"2021-07-23T05:03:18.806625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(np.transpose(train_data[0][0].numpy(), (1, 2, 0)))\ndef get_bounds(boxes):\n    bounds = str(boxes)\n    x = ast.literal_eval(re.search('({.+})', bounds).group(0))\n    df_box1 = pd.DataFrame(x[0])\n    #df_box2 = pd.DataFrame(x[1])\n    return df_box1#, df_box2","metadata":{"execution":{"iopub.status.busy":"2021-07-23T05:03:21.405623Z","iopub.execute_input":"2021-07-23T05:03:21.406006Z","iopub.status.idle":"2021-07-23T05:03:21.578593Z","shell.execute_reply.started":"2021-07-23T05:03:21.405973Z","shell.execute_reply":"2021-07-23T05:03:21.577362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Plotting bounding boxes","metadata":{}},{"cell_type":"code","source":"def get_bounds(boxes):\n    bounds = str(boxes)\n    x = ast.literal_eval(re.search('({.+})', bounds).group(0))\n    df_box1 = pd.DataFrame(x[0])\n    #df_box2 = pd.DataFrame(x[1])\n    return df_box1#, df_box2\n\nboxes = df_final.iloc[:1,:1]\nbounds = str(boxes)\nx = ast.literal_eval(re.search('({.+})', bounds).group(0))\nlengths = df_final.boxes.map(len(x))\npd.DataFrame({'x1':[x[0].get('x')],'y1':[x[0].get('y')],'w1':[x[0].get('width')],'h1':[x[0].get('height')],\n             'x2':[x[1].get('x')],'y2':[x[1].get('y')],'w2':[x[1].get('width')],'h2':[x[1].get('height')]})\n\ni = get_bounds(str(df_final.iloc[:1,:1]))\nget_bounds(df_final.boxes[0])\nx, y, w, h = get_bounds(df_final.boxes[0].values)\ndf_copy = df_final.copy()\ndf_boxes = df_copy.loc[df_copy.boxes.notna()]\ndf_boxes[['x','y','width','height']] = df_boxes.boxes.apply(get_bounds(bounds))","metadata":{"execution":{"iopub.status.busy":"2021-07-17T04:46:49.87728Z","iopub.execute_input":"2021-07-17T04:46:49.877624Z","iopub.status.idle":"2021-07-17T04:46:49.939218Z","shell.execute_reply.started":"2021-07-17T04:46:49.877587Z","shell.execute_reply":"2021-07-17T04:46:49.937221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Stratification/subset sampling\n\nhttps://stackoverflow.com/questions/50544730/how-do-i-split-a-custom-dataset-into-training-and-test-datasets\n\nhttps://pytorch.org/docs/master/data.html#torch.utils.data.SubsetRandomSampler","metadata":{}},{"cell_type":"markdown","source":"torch.utils.data.random_split(dataset, lengths, generator=<torch._C.Generator object>)[SOURCE]\n\nRandomly split a dataset into non-overlapping new datasets of given lengths. Optionally fix the generator for reproducible results, e.g.:\n\n>>> random_split(range(10), [3, 7], \ngenerator=torch.Generator().manual_seed(42))\n\nhttps://pytorch.org/docs/stable/data.html\n\n","metadata":{}},{"cell_type":"markdown","source":"# Transfer Learning (VGG19)","metadata":{}},{"cell_type":"code","source":"\"\"\"from tensorflow.keras.applications.vgg19 import VGG19\n\nbase_model = VGG19(input_shape=(224,224,3), \n                         include_top=False,\n                         weights=\"imagenet\")\"\"\"","metadata":{"execution":{"iopub.status.busy":"2021-07-17T04:46:49.94046Z","iopub.status.idle":"2021-07-17T04:46:49.941068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# torch_model = torch.hub.load('pytorch/vision:v0.9.0', 'vgg19_bn', pretrained=True)\n# model.eval()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T04:46:49.942603Z","iopub.status.idle":"2021-07-17T04:46:49.943184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"https://www.kaggle.com/givkashi/transfer-learning-with-vgg19","metadata":{}},{"cell_type":"code","source":"from torchvision import models","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:15:57.937237Z","iopub.execute_input":"2021-07-23T06:15:57.937629Z","iopub.status.idle":"2021-07-23T06:15:57.941418Z","shell.execute_reply.started":"2021-07-23T06:15:57.937594Z","shell.execute_reply":"2021-07-23T06:15:57.940461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vgg19 = models.vgg19(pretrained=True)","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:15:58.592421Z","iopub.execute_input":"2021-07-23T06:15:58.592779Z","iopub.status.idle":"2021-07-23T06:16:27.222785Z","shell.execute_reply.started":"2021-07-23T06:15:58.592747Z","shell.execute_reply":"2021-07-23T06:16:27.221944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vgg = vgg19","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:16:27.224373Z","iopub.execute_input":"2021-07-23T06:16:27.224716Z","iopub.status.idle":"2021-07-23T06:16:27.228293Z","shell.execute_reply.started":"2021-07-23T06:16:27.224678Z","shell.execute_reply":"2021-07-23T06:16:27.227488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for param in vgg.features.parameters():\n    print(len(param))","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2021-07-23T06:16:27.22992Z","iopub.execute_input":"2021-07-23T06:16:27.23041Z","iopub.status.idle":"2021-07-23T06:16:27.248049Z","shell.execute_reply.started":"2021-07-23T06:16:27.230373Z","shell.execute_reply":"2021-07-23T06:16:27.246949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Freeze training for all \"features\" layers\nfor param in vgg.features.parameters():\n    param.requires_grad = False","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:16:27.249492Z","iopub.execute_input":"2021-07-23T06:16:27.249743Z","iopub.status.idle":"2021-07-23T06:16:27.256986Z","shell.execute_reply.started":"2021-07-23T06:16:27.249719Z","shell.execute_reply":"2021-07-23T06:16:27.256082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vgg.classifier[-1]","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:16:27.258322Z","iopub.execute_input":"2021-07-23T06:16:27.258922Z","iopub.status.idle":"2021-07-23T06:16:27.265277Z","shell.execute_reply.started":"2021-07-23T06:16:27.258884Z","shell.execute_reply":"2021-07-23T06:16:27.264292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_inputs = vgg.classifier[-1].in_features","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:16:27.266802Z","iopub.execute_input":"2021-07-23T06:16:27.267457Z","iopub.status.idle":"2021-07-23T06:16:27.272194Z","shell.execute_reply.started":"2021-07-23T06:16:27.267417Z","shell.execute_reply":"2021-07-23T06:16:27.271282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.nn as nn\n\n# check if CUDA is available\ntrain_on_gpu = torch.cuda.is_available()\n\nif not train_on_gpu:\n    print('CUDA is not available.  Training on CPU ...')\nelse:\n    print('CUDA is available!  Training on GPU ...')\n    \n    \nn_inputs = vgg.classifier[-1].in_features\n\nfc_layer = nn.Linear(n_inputs, len(classes))\n\nvgg.classifier[-1] = fc_layer\n\n# after completing your model, if GPU is available, move the model to GPU\nif train_on_gpu:\n    vgg.cuda()\n    \nprint(vgg.classifier[6].out_features)\nprint(vgg)","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:16:27.273607Z","iopub.execute_input":"2021-07-23T06:16:27.274107Z","iopub.status.idle":"2021-07-23T06:16:31.797011Z","shell.execute_reply.started":"2021-07-23T06:16:27.274037Z","shell.execute_reply":"2021-07-23T06:16:31.794119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n# define dataloader parameters\nbatch_size = 19\nnum_workers=0\n\n# obtain one batch of training images\ndataiter = iter(train_loader)\nimages, labels = dataiter.next()\nprint(len(images))\nimages = images.numpy() # convert images to numpy for display\n\n# plot the images in the batch, along with the corresponding labels\nfig = plt.figure(figsize=(25, 4))\nfor idx in np.arange(19):\n    ax = fig.add_subplot(2, 20/2, idx+1, xticks=[], yticks=[])\n    plt.imshow(np.transpose(images[idx], (1, 2, 0)))\n    ax.set_title(labels[idx])","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:20:36.139449Z","iopub.execute_input":"2021-07-23T06:20:36.139774Z","iopub.status.idle":"2021-07-23T06:20:37.523272Z","shell.execute_reply.started":"2021-07-23T06:20:36.139745Z","shell.execute_reply":"2021-07-23T06:20:37.52242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.optim as optim\n\ncriterion = nn.CrossEntropyLoss()\noptimizer = optim.Adam(vgg.classifier.parameters(), lr=0.001)","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:20:42.562055Z","iopub.execute_input":"2021-07-23T06:20:42.562394Z","iopub.status.idle":"2021-07-23T06:20:42.568241Z","shell.execute_reply.started":"2021-07-23T06:20:42.562363Z","shell.execute_reply":"2021-07-23T06:20:42.567306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# number of epochs to train the model\nn_epochs = 2\n\n\nfor epoch in range(1, n_epochs+1):\n                   \n    train_loss= 0.0\n    \n    ###############\n    # model train #datatarget\n    ###############\n    \n    for batch, (image, label) in enumerate(train_loader, 1):\n        \n        image = image\n        label = label\n        if train_on_gpu:\n            image, label = image.cuda(), label.cuda()\n           \n        optimizer.zero_grad()\n        \n        output = vgg(image)\n        \n        loss = criterion(output, label)\n        \n        loss.backward()\n        \n        optimizer.step()\n        \n        train_loss += loss.item()\n        \n        if batch % 20 == 19:\n            \n            print(f'Epoch: {epoch} of {n_epochs}, Batch: {batch} of {len(train_loader)}, Loss: {loss.item():.4f}')\n            train_loss = 0.0","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:26:47.0618Z","iopub.execute_input":"2021-07-23T06:26:47.062179Z","iopub.status.idle":"2021-07-23T06:28:53.414422Z","shell.execute_reply.started":"2021-07-23T06:26:47.062149Z","shell.execute_reply":"2021-07-23T06:28:53.413531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_loader)/19","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:13:26.835426Z","iopub.execute_input":"2021-07-23T06:13:26.835874Z","iopub.status.idle":"2021-07-23T06:13:26.841364Z","shell.execute_reply.started":"2021-07-23T06:13:26.835836Z","shell.execute_reply":"2021-07-23T06:13:26.840588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Epoch: {epoch} of {n_epochs}, Batch: {batch} of {len(train_loader)}, modelLoss: {loss.item():.4f}')\n            train_loss = 0.0","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_loader.is_cuda()","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:37:56.448576Z","iopub.execute_input":"2021-07-23T06:37:56.44901Z","iopub.status.idle":"2021-07-23T06:37:56.463612Z","shell.execute_reply.started":"2021-07-23T06:37:56.448972Z","shell.execute_reply":"2021-07-23T06:37:56.462214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_loader","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:37:45.906498Z","iopub.execute_input":"2021-07-23T06:37:45.906828Z","iopub.status.idle":"2021-07-23T06:37:45.912074Z","shell.execute_reply.started":"2021-07-23T06:37:45.90679Z","shell.execute_reply":"2021-07-23T06:37:45.910864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test it out\ncorrect = 0\ntotal = 0\n# since we're not trainingmodel, we don't need to calculate the gradients for our outputs\nwith torch.no_grad():\n    for data in val_loader:\n        images, labels = data\n        # calculate outputs by running images through the network\n        outputs = vgg(images)\n        # the class with the highest energy is what we choose as prediction\n        _, predicted = torch.max(outputs.data, 1)\n        total += labels.size(0)\n        correct += (predicted == labels).sum().item()\n\nprint('Accuracy of the network on the test images: %d %%' % (\n    100 * correct / total))","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:37:10.937978Z","iopub.execute_input":"2021-07-23T06:37:10.938439Z","iopub.status.idle":"2021-07-23T06:37:11.160474Z","shell.execute_reply.started":"2021-07-23T06:37:10.938403Z","shell.execute_reply":"2021-07-23T06:37:11.157864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_total","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:49:44.814207Z","iopub.execute_input":"2021-07-23T06:49:44.814558Z","iopub.status.idle":"2021-07-23T06:49:44.819782Z","shell.execute_reply.started":"2021-07-23T06:49:44.814527Z","shell.execute_reply":"2021-07-23T06:49:44.81884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# track test loss \n\nval_loss = 0.0\nclass_correct = list(0. for i in range(4))\nclass_total = list(0. for i in range(4))\n\nvgg.eval() # eval mode\n\n# iterate over test data\nfor data, target in val_loader:\n    # move tensors to GPU if CUDA is available\n    if train_on_gpu:\n        data, target = data.cuda(), target.cuda()\n    # forward pass: compute predicted outputs by passing inputs to the model\n    output = vgg(data)\n    # calculate the batch loss\n    loss = criterion(output, target)\n    # update  test loss \n    val_loss += loss.item()*data.size(0)\n    \n    # convert output probabilities to predicted class\n    _, pred = torch.max(output, 1)    \n    # compare predictions to true label\n    correct_tensor = pred.eq(target.data.view_as(pred))\n    correct = np.squeeze(correct_tensor.numpy()) if not train_on_gpu else np.squeeze(correct_tensor.cpu().numpy())\n    # calculate test accuracy for each object class\n    \n    for i in range(14):\n        \n        label = target.data[i]\n        class_correct[label] += correct[i].item()\n        class_total[label] += 1\n\n# calculate avg test loss\nval_loss = val_loss/len(val_loader.dataset)\nprint('Test Loss: {:.6f}\\n'.format(test_loss))\n\nfor i in range(4):\n    if class_total[i] > 0:\n        print('Validation Accuracy of %5s: %2d%% (%2d/%2d)' % (\n            classes[i], 100 * class_correct[i] / class_total[i],\n            np.sum(class_correct[i]), np.sum(class_total[i])))\n    else:\n        print('Validation Accuracy of %5s: N/A (no training examples)' % (classes[i]))\n\nprint('\\Validation Accuracy (Overall): %2d%% (%2d/%2d)' % (\n    100. * np.sum(class_correct) / np.sum(class_total),\n    np.sum(class_correct), np.sum(class_total)))","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2021-07-23T06:51:05.858107Z","iopub.execute_input":"2021-07-23T06:51:05.858434Z","iopub.status.idle":"2021-07-23T06:51:17.55177Z","shell.execute_reply.started":"2021-07-23T06:51:05.858404Z","shell.execute_reply":"2021-07-23T06:51:17.550791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(val_loader)","metadata":{"execution":{"iopub.status.busy":"2021-07-23T06:44:38.270659Z","iopub.execute_input":"2021-07-23T06:44:38.27099Z","iopub.status.idle":"2021-07-23T06:44:38.278086Z","shell.execute_reply.started":"2021-07-23T06:44:38.270959Z","shell.execute_reply":"2021-07-23T06:44:38.2771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Use following to test instead of the above cell ","metadata":{}},{"cell_type":"code","source":"##DIP added ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test it out\ncorrect = 0\ntotal = 0\nprob_val = []\npred_val =[]\n\nmodel.eval()\n# since we're not training, we don't need to calculate the gradients for our outputs\nwith torch.no_grad():\n    for data in val_loader:\n        images, labels = data\n        #print(labels)\n        # calculate outputs by running images through the network\n        outputs = model(images)\n        # the class with the highest energy is what we choose as prediction\n        #print(outputs.data)\n        probability = torch.nn.functional.softmax(outputs.data,dim=1)\n        #print(probability)\n        prob_values, predicted = torch.max(probability, 1)\n        #print(predicted.numpy())\n        #print(prob_values)\n        #print(labels.size(0))\n        [pred_val.append(x ) for x in predicted.numpy()]\n        [prob_val.append(x) for x in prob_values.numpy()]\n        total += labels.size(0)\n        correct += (predicted == labels).sum().item()\n        #print(correct)\n\nprint('Accuracy of the network on the test images: %d %%' % (\n    100 * correct / total))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#predicted labels with confidence scores \npd.DataFrame(list(zip(pred_val, prob_val)) , columns =[\"Class\" , \"Confidence\"])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}