{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Dicom to PNG Necessary Tool","metadata":{}},{"cell_type":"code","source":"!conda install '/kaggle/input/pydicom-conda-helper/libjpeg-turbo-2.1.0-h7f98852_0.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/libgcc-ng-9.3.0-h2828fa1_19.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/gdcm-2.8.9-py37h500ead1_1.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/conda-4.10.1-py37h89c1867_0.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/certifi-2020.12.5-py37h89c1867_1.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/openssl-1.1.1k-h7f98852_0.tar.bz2' -c conda-forge -y","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-18T12:32:40.080366Z","iopub.execute_input":"2022-12-18T12:32:40.080744Z","iopub.status.idle":"2022-12-18T12:33:26.894303Z","shell.execute_reply.started":"2022-12-18T12:32:40.080709Z","shell.execute_reply":"2022-12-18T12:33:26.893284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom PIL import Image\nimport pandas as pd\nfrom tqdm.auto import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-18T12:33:26.897053Z","iopub.execute_input":"2022-12-18T12:33:26.897614Z","iopub.status.idle":"2022-12-18T12:33:26.902302Z","shell.execute_reply.started":"2022-12-18T12:33:26.897566Z","shell.execute_reply":"2022-12-18T12:33:26.901433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\ndef read_xray(path, voi_lut = True, fix_monochrome = True):\n    dicom = pydicom.read_file(path)\n    data = apply_voi_lut(dicom.pixel_array, dicom)\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data  \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n\ndef resize(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n    im = Image.fromarray(array)\n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        im = im.resize((size, size), resample)\n    return im","metadata":{"execution":{"iopub.status.busy":"2022-12-18T12:33:26.904378Z","iopub.execute_input":"2022-12-18T12:33:26.904754Z","iopub.status.idle":"2022-12-18T12:33:26.913790Z","shell.execute_reply.started":"2022-12-18T12:33:26.904710Z","shell.execute_reply":"2022-12-18T12:33:26.912934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# .dcm to .png","metadata":{}},{"cell_type":"code","source":"split = 'test'\nsave_dir = f'/kaggle/working/{split}/'\nos.makedirs(save_dir, exist_ok=True)\nsave_dir = f'/kaggle/working/{split}/study/'\nos.makedirs(save_dir, exist_ok=True)\ncount = 0\nfor dirname, _, filenames in tqdm(os.walk(f'../input/siim-covid19-detection/{split}')):\n    for file in filenames:\n        # set keep_ratio=True to have original aspect ratio\n        xray = read_xray(os.path.join(dirname, file))\n        im = resize(xray, size=224)  \n        study = dirname.split('/')[-2] + '_study.png'\n        im.save(os.path.join(save_dir, study))\n        del im\n        if (count > 1000):\n            break\n    if (count > 1000):\n        break","metadata":{"execution":{"iopub.status.busy":"2022-12-18T12:33:26.915428Z","iopub.execute_input":"2022-12-18T12:33:26.915836Z","iopub.status.idle":"2022-12-18T12:40:15.724343Z","shell.execute_reply.started":"2022-12-18T12:33:26.915801Z","shell.execute_reply":"2022-12-18T12:40:15.723527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"split = 'train'\nsave_dir = f'/kaggle/working/{split}/'\nos.makedirs(save_dir, exist_ok=True)\nsave_dir = f'/kaggle/working/{split}/study/'\nos.makedirs(save_dir, exist_ok=True)\n\ncount = 0\nfor dirname, _, filenames in tqdm(os.walk(f'../input/siim-covid19-detection/{split}')):\n    for file in filenames:\n        # set keep_ratio=True to have original aspect ratio\n        xray = read_xray(os.path.join(dirname, file))\n        im = resize(xray, size=224)  \n        study = dirname.split('/')[-2] + '_study.png'\n        im.save(os.path.join(save_dir, study))\n        del im\n        if (count > 1000):\n            break\n    if (count > 1000):\n        break","metadata":{"execution":{"iopub.status.busy":"2022-12-18T12:40:15.725682Z","iopub.execute_input":"2022-12-18T12:40:15.726203Z","iopub.status.idle":"2022-12-18T12:47:14.096432Z","shell.execute_reply.started":"2022-12-18T12:40:15.726163Z","shell.execute_reply":"2022-12-18T12:47:14.095556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## PNG and Label","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('../input/siim-covid19-detection/train_study_level.csv')\nsub_df = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-12-18T12:47:14.097752Z","iopub.execute_input":"2022-12-18T12:47:14.098268Z","iopub.status.idle":"2022-12-18T12:47:14.122585Z","shell.execute_reply.started":"2022-12-18T12:47:14.098226Z","shell.execute_reply":"2022-12-18T12:47:14.121915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt \n\ncount = 0\nfor dirname, _, filenames in os.walk(\"/kaggle/working/train/study/\"):\n    for file in filenames:\n        print(dirname+\"/\"+file)\n        image = cv2.imread(dirname+\"/\"+file)\n        is_negative = train_df[train_df.id == file[:-4]][\"Negative for Pneumonia\"]\n        is_typical = train_df[train_df.id == file[:-4]][\"Typical Appearance\"]\n        is_indeterminate = train_df[train_df.id == file[:-4]][\"Indeterminate Appearance\"]\n        is_atypical = train_df[train_df.id == file[:-4]][\"Atypical Appearance\"]\n        correct_label = [is_negative.values[0], is_typical.values[0], is_indeterminate.values[0], is_atypical.values[0]]\n        print(\"correct_label = \", correct_label)\n        plt.imshow(image)\n        plt.show()\n        print(image.shape)\n        count += 1\n        if (count == 1):\n            break","metadata":{"execution":{"iopub.status.busy":"2022-12-18T12:47:14.123853Z","iopub.execute_input":"2022-12-18T12:47:14.124193Z","iopub.status.idle":"2022-12-18T12:47:14.297028Z","shell.execute_reply.started":"2022-12-18T12:47:14.124156Z","shell.execute_reply":"2022-12-18T12:47:14.296178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# learn study","metadata":{}},{"cell_type":"code","source":"import torch\nclass COVID_Datasets(torch.utils.data.Dataset):\n    def __init__(self):\n        self.images = []\n        self.labels = []\n        self.filenames = []\n    def __len__(self):\n        return len(self.images)\n    def add_data(self, image, label, filename):\n        image = (image / 255.0).astype(np.float32).transpose((2, 0, 1))\n        label = (np.array([label]).reshape(4))\n        self.images.append(image)\n        self.labels.append(label)\n        self.filenames.append(filename)\n    def get_filename(self, idx):\n        return self.filenames[idx]\n    def __getitem__(self, idx):\n        return self.images[idx], self.labels[idx]","metadata":{"execution":{"iopub.status.busy":"2022-12-18T12:47:14.299310Z","iopub.execute_input":"2022-12-18T12:47:14.299818Z","iopub.status.idle":"2022-12-18T12:47:14.307286Z","shell.execute_reply.started":"2022-12-18T12:47:14.299780Z","shell.execute_reply":"2022-12-18T12:47:14.306425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = COVID_Datasets()\ncount = 0\nfor dirname, _, filenames in os.walk(\"/kaggle/working/train/study/\"):\n    for file in filenames:\n        image = cv2.imread(dirname+\"/\"+file)\n        is_negative = train_df[train_df.id == file[:-4]][\"Negative for Pneumonia\"]\n        is_typical = train_df[train_df.id == file[:-4]][\"Typical Appearance\"]\n        is_indeterminate = train_df[train_df.id == file[:-4]][\"Indeterminate Appearance\"]\n        is_atypical = train_df[train_df.id == file[:-4]][\"Atypical Appearance\"]\n        correct_label = [is_negative.values[0], is_typical.values[0], is_indeterminate.values[0], is_atypical.values[0]]\n        if (count == 0):\n            print(dirname+\"/\"+file)\n            print(\"correct_label = \", correct_label)\n            plt.imshow(image)\n            plt.show()\n            print(image.shape)\n        count += 1\n        train_dataset.add_data(image, correct_label, file)\ntrainloader = torch.utils.data.DataLoader(train_dataset, batch_size = 16, shuffle = True, num_workers = 4)","metadata":{"execution":{"iopub.status.busy":"2022-12-18T12:47:14.308765Z","iopub.execute_input":"2022-12-18T12:47:14.309116Z","iopub.status.idle":"2022-12-18T12:47:29.545065Z","shell.execute_reply.started":"2022-12-18T12:47:14.309080Z","shell.execute_reply":"2022-12-18T12:47:29.543914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"import torchvision\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nmodel = torch.load(\"/kaggle/input/vgg-pretrained/vgg16-pretrained.pth\")\nmodel = nn.Sequential(\n    model,\n    nn.Linear(1000, 4),\n    nn.Softmax()\n).to(\"cuda\")","metadata":{"execution":{"iopub.status.busy":"2022-12-19T05:41:23.479501Z","iopub.execute_input":"2022-12-19T05:41:23.479849Z","iopub.status.idle":"2022-12-19T05:41:32.354795Z","shell.execute_reply.started":"2022-12-19T05:41:23.479814Z","shell.execute_reply":"2022-12-19T05:41:32.353929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#criterion = nn.CrossEntropyLoss()\ncriterion = nn.MSELoss()\n\noptimizer = optim.SGD(model.parameters(), lr=0.001)","metadata":{"execution":{"iopub.status.busy":"2022-12-18T12:47:35.019143Z","iopub.execute_input":"2022-12-18T12:47:35.019465Z","iopub.status.idle":"2022-12-18T12:47:35.024298Z","shell.execute_reply.started":"2022-12-18T12:47:35.019428Z","shell.execute_reply":"2022-12-18T12:47:35.023417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt \n\nmax_epoch = 51\nlosses = []\nfor epoch in range(max_epoch):\n    for i , (images, labels) in enumerate(trainloader):\n        images = images.float().to(\"cuda\")\n        labels = labels.float().to(\"cuda\")\n        outputs = model(images)\n        optimizer.zero_grad() \n        loss = criterion(outputs, labels.float())\n        loss.backward()\n        optimizer.step()\n        losses.append(loss.item())\n    if (epoch % 10 == 0):\n        plt.plot(losses)\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-18T12:47:35.025561Z","iopub.execute_input":"2022-12-18T12:47:35.026163Z","iopub.status.idle":"2022-12-18T13:22:45.947791Z","shell.execute_reply.started":"2022-12-18T12:47:35.026116Z","shell.execute_reply":"2022-12-18T13:22:45.946605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# study predict","metadata":{}},{"cell_type":"markdown","source":"# study string","metadata":{}},{"cell_type":"code","source":"test_dataset = COVID_Datasets()\ncount = 0\nfor dirname, _, filenames in os.walk(\"/kaggle/working/test/study/\"):\n    for file in filenames:\n        image = cv2.imread(dirname+\"/\"+file)\n        correct_label = [0,0,0,0]\n        test_dataset.add_data(image, correct_label, file)","metadata":{"execution":{"iopub.status.busy":"2022-12-18T13:50:02.585791Z","iopub.execute_input":"2022-12-18T13:50:02.586119Z","iopub.status.idle":"2022-12-18T13:50:11.414748Z","shell.execute_reply.started":"2022-12-18T13:50:02.586089Z","shell.execute_reply":"2022-12-18T13:50:11.413854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset.__getitem__(0)\ntest_dataset.get_filename(0)[:-4]","metadata":{"execution":{"iopub.status.busy":"2022-12-18T13:50:11.416457Z","iopub.execute_input":"2022-12-18T13:50:11.416884Z","iopub.status.idle":"2022-12-18T13:50:11.423284Z","shell.execute_reply.started":"2022-12-18T13:50:11.416837Z","shell.execute_reply":"2022-12-18T13:50:11.422308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(test_dataset)):\n    test_image = test_dataset.__getitem__(i)[0]\n    #print(test_image.shape)\n    test_image = torch.tensor([test_image])#.detach().float()\n    output = model(test_image.to(\"cuda\")).cpu().detach().numpy()\n    print(output)\n    pred = np.argmax(output)\n    if (pred == 0) :\n        prediction_string = \"negative 1 0 0 1 1\"\n    elif (pred == 1) :\n        prediction_string = \"typical 1 0 0 1 1\"\n    elif (pred == 2) :\n        prediction_string = \"indeterminate 1 0 0 1 1\"\n    elif (pred == 3) :\n        prediction_string = \"atypical 1 0 0 1 1\"\n    else :\n        prediction_string = \"none 1 0 0 1 1\"\n    \n    sub_df.loc[sub_df[sub_df.id==test_dataset.get_filename(i)[:-4]].index[0]][\"PredictionString\"] = prediction_string\n    print(i, prediction_string)","metadata":{"execution":{"iopub.status.busy":"2022-12-18T13:50:11.425225Z","iopub.execute_input":"2022-12-18T13:50:11.425925Z","iopub.status.idle":"2022-12-18T13:52:38.623777Z","shell.execute_reply.started":"2022-12-18T13:50:11.425756Z","shell.execute_reply":"2022-12-18T13:52:38.622917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-18T13:52:38.625107Z","iopub.execute_input":"2022-12-18T13:52:38.625454Z","iopub.status.idle":"2022-12-18T13:52:38.635118Z","shell.execute_reply.started":"2022-12-18T13:52:38.625418Z","shell.execute_reply":"2022-12-18T13:52:38.634237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df[sub_df.PredictionString == \"indeterminate 1 0 0 1 1\"]","metadata":{"execution":{"iopub.status.busy":"2022-12-18T13:52:48.520370Z","iopub.execute_input":"2022-12-18T13:52:48.520745Z","iopub.status.idle":"2022-12-18T13:52:48.533895Z","shell.execute_reply.started":"2022-12-18T13:52:48.520710Z","shell.execute_reply":"2022-12-18T13:52:48.532859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df[sub_df.PredictionString == \"typical 1 0 0 1 1\"]","metadata":{"execution":{"iopub.status.busy":"2022-12-18T13:52:48.968726Z","iopub.execute_input":"2022-12-18T13:52:48.969009Z","iopub.status.idle":"2022-12-18T13:52:48.981396Z","shell.execute_reply.started":"2022-12-18T13:52:48.968983Z","shell.execute_reply":"2022-12-18T13:52:48.980467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df[sub_df.PredictionString == \"atypical 1 0 0 1 1\"]","metadata":{"execution":{"iopub.status.busy":"2022-12-18T13:52:49.499819Z","iopub.execute_input":"2022-12-18T13:52:49.500087Z","iopub.status.idle":"2022-12-18T13:52:49.510426Z","shell.execute_reply.started":"2022-12-18T13:52:49.500060Z","shell.execute_reply":"2022-12-18T13:52:49.509534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-18T13:52:50.903215Z","iopub.execute_input":"2022-12-18T13:52:50.903527Z","iopub.status.idle":"2022-12-18T13:52:51.108398Z","shell.execute_reply.started":"2022-12-18T13:52:50.903497Z","shell.execute_reply":"2022-12-18T13:52:51.107606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}