{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"! conda install -c conda-forge gdcm -y","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-25T10:13:56.511649Z","iopub.execute_input":"2021-07-25T10:13:56.512218Z","iopub.status.idle":"2021-07-25T10:15:10.045136Z","shell.execute_reply.started":"2021-07-25T10:13:56.512093Z","shell.execute_reply":"2021-07-25T10:15:10.044111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport pydicom\nimport glob\nfrom tqdm import tqdm\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport matplotlib.pyplot as plt\nfrom skimage import exposure\nimport cv2\nimport warnings\nfrom fastai.vision.all import *\nfrom fastai.medical.imaging import *\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2021-07-25T10:15:10.04693Z","iopub.execute_input":"2021-07-25T10:15:10.047342Z","iopub.status.idle":"2021-07-25T10:15:13.747344Z","shell.execute_reply.started":"2021-07-25T10:15:10.047306Z","shell.execute_reply":"2021-07-25T10:15:13.746132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_path = Path('../input/siim-covid19-detection')","metadata":{"execution":{"iopub.status.busy":"2021-07-25T10:15:13.74944Z","iopub.execute_input":"2021-07-25T10:15:13.749763Z","iopub.status.idle":"2021-07-25T10:15:13.754328Z","shell.execute_reply.started":"2021-07-25T10:15:13.749733Z","shell.execute_reply":"2021-07-25T10:15:13.753295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_path.ls()","metadata":{"execution":{"iopub.status.busy":"2021-07-25T10:15:13.756023Z","iopub.execute_input":"2021-07-25T10:15:13.756533Z","iopub.status.idle":"2021-07-25T10:15:13.775778Z","shell.execute_reply.started":"2021-07-25T10:15:13.756478Z","shell.execute_reply":"2021-07-25T10:15:13.774222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_study_df = pd.read_csv(dataset_path/'train_study_level.csv')\ntrain_study_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-25T10:15:13.78039Z","iopub.execute_input":"2021-07-25T10:15:13.781361Z","iopub.status.idle":"2021-07-25T10:15:13.828794Z","shell.execute_reply.started":"2021-07-25T10:15:13.781301Z","shell.execute_reply":"2021-07-25T10:15:13.82773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_classes = ['Negative for Pneumonia', 'Typical Appearance', 'Indeterminate Appearance', 'Atypical Appearance']\nnp.unique(train_study_df[study_classes].values, axis=0)","metadata":{"execution":{"iopub.status.busy":"2021-07-25T10:15:13.83041Z","iopub.execute_input":"2021-07-25T10:15:13.830758Z","iopub.status.idle":"2021-07-25T10:15:13.856866Z","shell.execute_reply.started":"2021-07-25T10:15:13.830724Z","shell.execute_reply":"2021-07-25T10:15:13.855921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_study_df[study_classes].values.sum(axis=0)","metadata":{"execution":{"iopub.status.busy":"2021-07-25T10:15:13.85822Z","iopub.execute_input":"2021-07-25T10:15:13.858714Z","iopub.status.idle":"2021-07-25T10:15:13.8723Z","shell.execute_reply.started":"2021-07-25T10:15:13.858667Z","shell.execute_reply":"2021-07-25T10:15:13.871139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,5))\nplt.bar([1,2,3,4], train_study_df[study_classes].values.sum(axis=0))\nplt.xticks([1,2,3,4], study_classes)\nplt.ylabel('Frequency')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-25T10:15:13.875307Z","iopub.execute_input":"2021-07-25T10:15:13.875644Z","iopub.status.idle":"2021-07-25T10:15:14.044788Z","shell.execute_reply.started":"2021-07-25T10:15:13.875613Z","shell.execute_reply":"2021-07-25T10:15:14.043361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image_df = pd.read_csv(dataset_path/'train_image_level.csv')\ntrain_image_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-25T10:15:14.046579Z","iopub.execute_input":"2021-07-25T10:15:14.046902Z","iopub.status.idle":"2021-07-25T10:15:14.104074Z","shell.execute_reply.started":"2021-07-25T10:15:14.046869Z","shell.execute_reply":"2021-07-25T10:15:14.102915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image_df['split_label'] = train_image_df.label.apply(lambda x: [x.split()[offs:offs+6] for offs in range(0, len(x.split()), 6)])","metadata":{"execution":{"iopub.status.busy":"2021-07-25T10:15:14.105774Z","iopub.execute_input":"2021-07-25T10:15:14.10625Z","iopub.status.idle":"2021-07-25T10:15:14.140228Z","shell.execute_reply.started":"2021-07-25T10:15:14.1062Z","shell.execute_reply":"2021-07-25T10:15:14.138705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classes_freq = []\nfor i in range(len(train_image_df)):\n    for j in train_image_df.iloc[i].split_label:\n        classes_freq.append(j[0])\nplt.hist(classes_freq)\nplt.ylabel('Frequency')","metadata":{"execution":{"iopub.status.busy":"2021-07-25T10:15:14.141898Z","iopub.execute_input":"2021-07-25T10:15:14.14225Z","iopub.status.idle":"2021-07-25T10:15:15.049327Z","shell.execute_reply.started":"2021-07-25T10:15:14.142218Z","shell.execute_reply":"2021-07-25T10:15:15.048221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bbox_areas = []\nfor i in range(len(train_image_df)):\n    for j in train_image_df.iloc[i].split_label:\n        bbox_areas.append((float(j[4]) - float(j[2])) * (float(j[5])-float(j[3])))\nplt.hist(bbox_areas)\nplt.ylabel('Frequency')","metadata":{"execution":{"iopub.status.busy":"2021-07-25T10:15:15.050793Z","iopub.execute_input":"2021-07-25T10:15:15.051165Z","iopub.status.idle":"2021-07-25T10:15:15.982906Z","shell.execute_reply.started":"2021-07-25T10:15:15.051133Z","shell.execute_reply":"2021-07-25T10:15:15.981651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def dicom2array(path, voi_lut=True, fix_monochrome=True):\n    dicom = pydicom.read_file(path)\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n    if fix_monochrome and dicom.PhotometricInterpretation == 'MONOCHROME1':\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    out_im = Image.fromarray(data)\n    out_im.save(f'/kaggle/working/{path.stem}.png')\n\ndef plot_img(img, size=(7,7), is_rgb=True, title=\"\", cmap='gray'):\n    plt.figure(figsize=size)\n    plt.imshow(img, cmap=cmap)\n    plt.suptitle(title)\n    plt.show()\n\ndef plot_imgs(imgs, cols=4, size=7, is_rgb=True, title=\"\", cmap='gray', img_size=(500,500)):\n    rows = len(imgs)//cols + 1\n    fig = plt.figure(figsize=(cols*size, rows*size))\n    for i, img in enumerate(imgs):\n        if img_size is not None:\n            img = cv2.resize(img, img_size)\n        fig.add_subplot(rows,cols,i+1)\n        plt.imshow(img, cmap=cmap)\n    plt.suptitle(title)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-25T10:15:15.984415Z","iopub.execute_input":"2021-07-25T10:15:15.984735Z","iopub.status.idle":"2021-07-25T10:15:16.19714Z","shell.execute_reply.started":"2021-07-25T10:15:15.984705Z","shell.execute_reply":"2021-07-25T10:15:16.195884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dicom_paths = get_dicom_files(dataset_path/'train')","metadata":{"execution":{"iopub.status.busy":"2021-07-25T10:15:16.198842Z","iopub.execute_input":"2021-07-25T10:15:16.19921Z","iopub.status.idle":"2021-07-25T10:15:43.278646Z","shell.execute_reply.started":"2021-07-25T10:15:16.199172Z","shell.execute_reply":"2021-07-25T10:15:43.277308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for path in tqdm(dicom_paths[:1000]):\n    dicom2array(path)","metadata":{"execution":{"iopub.status.busy":"2021-07-25T10:15:43.280185Z","iopub.execute_input":"2021-07-25T10:15:43.280486Z","iopub.status.idle":"2021-07-25T10:54:09.489898Z","shell.execute_reply.started":"2021-07-25T10:15:43.280457Z","shell.execute_reply":"2021-07-25T10:54:09.487417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import zipfile\nziph = zipfile.ZipFile(f'/kaggle/working/train_1.zip', 'w', zipfile.ZIP_DEFLATED)\n\nfor root, dirs, files in os.walk(\"/kaggle/working/\"):\n    for file in tqdm(files):\n        file_path = os.path.join(root, file)\n        ziph.write(file_path)\n        os.remove(file_path)\n    ziph.close()","metadata":{"execution":{"iopub.status.busy":"2021-07-25T10:54:09.493562Z","iopub.execute_input":"2021-07-25T10:54:09.49398Z","iopub.status.idle":"2021-07-25T11:07:54.812969Z","shell.execute_reply.started":"2021-07-25T10:54:09.49393Z","shell.execute_reply":"2021-07-25T11:07:54.810564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(files)","metadata":{"execution":{"iopub.status.busy":"2021-07-24T21:23:33.260069Z","iopub.execute_input":"2021-07-24T21:23:33.260446Z","iopub.status.idle":"2021-07-24T21:23:33.268528Z","shell.execute_reply.started":"2021-07-24T21:23:33.260416Z","shell.execute_reply":"2021-07-24T21:23:33.267527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-07-24T21:21:52.777249Z","iopub.execute_input":"2021-07-24T21:21:52.777551Z","iopub.status.idle":"2021-07-24T21:21:52.781831Z","shell.execute_reply.started":"2021-07-24T21:21:52.777524Z","shell.execute_reply":"2021-07-24T21:21:52.780606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_path = \"/kaggle/working/d51cadde8626.png\"\nziph.write(file_path)\nos.remove(file_path)","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:26:01.929032Z","iopub.execute_input":"2021-07-24T18:26:01.929373Z","iopub.status.idle":"2021-07-24T18:26:02.087768Z","shell.execute_reply.started":"2021-07-24T18:26:01.929344Z","shell.execute_reply":"2021-07-24T18:26:02.086829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip -r /kaggle/working/ *","metadata":{"execution":{"iopub.status.busy":"2021-07-24T17:55:32.24684Z","iopub.execute_input":"2021-07-24T17:55:32.2473Z","iopub.status.idle":"2021-07-24T17:55:33.296219Z","shell.execute_reply.started":"2021-07-24T17:55:32.247258Z","shell.execute_reply":"2021-07-24T17:55:33.295008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\ndicom2array(dicom_paths[4])\ndicom_paths[4]","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:25:47.460779Z","iopub.execute_input":"2021-07-24T18:25:47.461266Z","iopub.status.idle":"2021-07-24T18:25:50.367788Z","shell.execute_reply.started":"2021-07-24T18:25:47.461224Z","shell.execute_reply":"2021-07-24T18:25:50.366962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom tqdm import tqdm\nfor path in tqdm(dicom_paths):\n    dicom2array(path) \n#plot_imgs(imgs)","metadata":{"execution":{"iopub.status.busy":"2021-07-24T17:27:33.824566Z","iopub.execute_input":"2021-07-24T17:27:33.824895Z","iopub.status.idle":"2021-07-24T17:28:54.838518Z","shell.execute_reply.started":"2021-07-24T17:27:33.824864Z","shell.execute_reply":"2021-07-24T17:28:54.836655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dicom_paths[0].stem","metadata":{"execution":{"iopub.status.busy":"2021-07-24T17:19:49.680268Z","iopub.execute_input":"2021-07-24T17:19:49.680589Z","iopub.status.idle":"2021-07-24T17:19:49.688766Z","shell.execute_reply.started":"2021-07-24T17:19:49.680559Z","shell.execute_reply":"2021-07-24T17:19:49.687921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_images_per_study = []\nfor i in (dataset_path/'train').ls():\n    num_images_per_study.append(len(get_dicom_files(i)))\n    if len(get_dicom_files(i)) > 5:\n        print(f'Study {i} had {len(get_dicom_files(i))} images')","metadata":{"execution":{"iopub.status.busy":"2021-07-19T13:50:23.631998Z","iopub.execute_input":"2021-07-19T13:50:23.632517Z","iopub.status.idle":"2021-07-19T13:50:36.500474Z","shell.execute_reply.started":"2021-07-19T13:50:23.63246Z","shell.execute_reply":"2021-07-19T13:50:36.499346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(num_images_per_study)","metadata":{"execution":{"iopub.status.busy":"2021-07-19T13:50:36.506202Z","iopub.execute_input":"2021-07-19T13:50:36.509105Z","iopub.status.idle":"2021-07-19T13:50:36.840481Z","shell.execute_reply.started":"2021-07-19T13:50:36.509043Z","shell.execute_reply":"2021-07-19T13:50:36.839378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def image_path(row):\n    study_path = dataset_path/'train'/row.StudyInstanceUID\n    for i in get_dicom_files(study_path):\n        if row.id.split('_')[0] == i.stem:\n            return i \n        \ntrain_image_df['image_path'] = train_image_df.apply(image_path, axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-07-19T13:50:36.843975Z","iopub.execute_input":"2021-07-19T13:50:36.844297Z","iopub.status.idle":"2021-07-19T13:50:44.311312Z","shell.execute_reply.started":"2021-07-19T13:50:36.844266Z","shell.execute_reply":"2021-07-19T13:50:44.310231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs = []\nimage_paths = train_image_df['image_path'].values\n\nthickness = 10\nscale = 5\n\nfor i  in range(8):\n    image_path = random.choice(image_paths)\n    print(image_path)\n    img = dicom2array(image_path)\n    img = cv2.resize(img, None, fx=1/scale, fy=1/scale)\n    img = np.stack([img, img, img], axis=-1)\n    for i in train_image_df.loc[train_image_df['image_path'] == image_path].split_label.values[0]:\n        if i[0] == 'opacity':\n            img = cv2.rectangle(img, (int(float(i[2])/scale), int(float(i[3])/scale)),\n                                     (int(float(i[4])/scale), int(float(i[5])/scale)),\n                                     [255,0,0], thickness)\n    img = cv2.resize(img, (500,500))\n    imgs.append(img)\nplot_imgs(imgs, cmap=None)","metadata":{"execution":{"iopub.status.busy":"2021-07-19T13:50:44.313642Z","iopub.execute_input":"2021-07-19T13:50:44.314039Z","iopub.status.idle":"2021-07-19T13:50:48.821211Z","shell.execute_reply.started":"2021-07-19T13:50:44.314009Z","shell.execute_reply":"2021-07-19T13:50:48.819705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}