{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport pydicom\nfrom glob import glob\nfrom tqdm.notebook import tqdm\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport matplotlib.pyplot as plt\nfrom skimage import exposure\nimport cv2\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":0.720041,"end_time":"2021-01-01T01:51:44.426457","exception":false,"start_time":"2021-01-01T01:51:43.706416","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_dir = '../input/siim-covid19-detection'","metadata":{"papermill":{"duration":0.026418,"end_time":"2021-01-01T01:51:44.508062","exception":false,"start_time":"2021-01-01T01:51:44.481644","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def dicom2array(path, voi_lut=True, fix_monochrome=True):\n    dicom = pydicom.read_file(path)\n    #VOI LUT Volume Of Interest Look Up Table\n    # VOI LUT (if available by DICOM device) is used to\n    # transform raw DICOM data to \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n        \n    \ndef plot_img(img, size=(7, 7), is_rgb=True, title=\"\", cmap='gray'):\n    plt.figure(figsize=size)\n    plt.imshow(img, cmap=cmap)\n    plt.suptitle(title)\n    plt.show()\n\n\ndef plot_imgs(imgs, cols=4, size=7, is_rgb=True, title=\"\", cmap='gray', img_size=(500,500)):\n    rows = len(imgs)//cols + 1\n    fig = plt.figure(figsize=(cols*size, rows*size))\n    for i, img in enumerate(imgs):\n        if img_size is not None:\n            img = cv2.resize(img, img_size)\n        fig.add_subplot(rows, cols, i+1)\n        plt.imshow(img, cmap=cmap)\n    plt.suptitle(title)\n    plt.show()\n","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","papermill":{"duration":0.040531,"end_time":"2021-01-01T01:51:44.567233","exception":false,"start_time":"2021-01-01T01:51:44.526702","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# simple plot","metadata":{}},{"cell_type":"code","source":"dicom_paths = glob(f'{dataset_dir}/train/*/*/*.dcm')\n\nimgs = [dicom2array(path) for path in dicom_paths[:4]]\nplot_imgs(imgs)","metadata":{"papermill":{"duration":9.946147,"end_time":"2021-01-01T01:51:54.532001","exception":false,"start_time":"2021-01-01T01:51:44.585854","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print(dicom_paths[:20])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Maybe, you can try some preprocess like equalize histogram. You can see the difference between before and after","metadata":{"papermill":{"duration":0.029261,"end_time":"2021-01-01T01:51:54.592507","exception":false,"start_time":"2021-01-01T01:51:54.563246","status":"completed"},"tags":[]}},{"cell_type":"code","source":"imgs = [exposure.equalize_hist(img) for img in imgs]\nplot_imgs(imgs)","metadata":{"papermill":{"duration":2.163571,"end_time":"2021-01-01T01:51:56.7856","exception":false,"start_time":"2021-01-01T01:51:54.622029","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# load & preprocess","metadata":{}},{"cell_type":"markdown","source":"Bokeh is a Python library for creating interactive visualizations for modern web browsers. It helps you build beautiful graphics, ranging from simple plots to complex dashboards with streaming datasets. With Bokeh, you can create JavaScript-powered visualizations without writing any JavaScript yourself.\n","metadata":{}},{"cell_type":"code","source":"from bokeh.plotting import figure as bokeh_figure\nfrom bokeh.io import output_notebook, show, output_file\nfrom bokeh.models import ColumnDataSource, HoverTool, Panel\nfrom bokeh.models.widgets import Tabs\nimport pandas as pd\nfrom PIL import Image\nfrom sklearn import preprocessing\nimport random\nfrom random import randint","metadata":{"papermill":{"duration":1.37697,"end_time":"2021-01-01T01:51:58.374581","exception":false,"start_time":"2021-01-01T01:51:56.997611","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(f'{dataset_dir}/train_image_level.csv')\ntrain_study = pd.read_csv(f'{dataset_dir}/train_study_level.csv')\ntrain.head()","metadata":{"papermill":{"duration":1.675464,"end_time":"2021-01-01T01:52:00.092785","exception":false,"start_time":"2021-01-01T01:51:58.417321","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_study.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# merge study csv\ntrain_study['StudyInstanceUID'] = train_study['id'].apply(lambda x: x.replace('_study', ''))\ndel train_study['id']\ntrain = train.merge(train_study, on='StudyInstanceUID')\nprint(train.shape)\ntrain.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# add StudyInstanceUID_count column\ngroup_col = 'StudyInstanceUID'\ndf=pd.DataFrame(train.groupby(group_col)['id'].count())\ndf.columns = [f'{group_col}_count']\ntrain=train.merge(df.reset_index(), on=group_col)\none_study_multi_image_df = train[train[f'{group_col}_count'] > 1]\nprint(len(one_study_multi_image_df))\ntrain = train[train[f'{group_col}_count'] == 1] # delete 'StudyInstanceUID_count > 1' data\none_study_multi_image_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"we have 512images with 'StudyInstanceUID_count > 1'.  \nSince the problem has not been solved, I deleted this data.  \nhttps://www.kaggle.com/c/siim-covid19-detection/discussion/239980","metadata":{}},{"cell_type":"code","source":"def bar_plot(train_df, variable):\n    \"\"\"\n        input: variable ex: \"Sex\"\n        output: bar plot & value count\n    \"\"\"\n    # get feature\n    var = train_df[variable]\n    # count number of categorical variable(value/sample)\n    varValue = var.value_counts()\n    \n    # visualize\n    plt.figure(figsize = (9,3))\n    plt.bar(varValue.index, varValue)\n    plt.xticks(varValue.index, varValue.index.values)\n    plt.ylabel(\"Frequency\")\n    plt.title(variable)\n    plt.show()\n    print(\"{}: \\n {}\".format(variable,varValue))\n    \ntrain['target'] = 'Negative for Pneumonia'\ntrain.loc[train['Typical Appearance']==1, 'target'] = 'Typical Appearance'\ntrain.loc[train['Indeterminate Appearance']==1, 'target'] = 'Indeterminate Appearance'\ntrain.loc[train['Atypical Appearance']==1, 'target'] = 'Atypical Appearance'\nbar_plot(train, 'target')    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.boxes.values[0] # x_min, y_min, width, height","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.label.values[0] # x_min, y_min, x_max, y_max","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train[~train.boxes.isnull()] \nclass_names = ['Typical Appearance', 'Indeterminate Appearance', 'Atypical Appearance'] # we have 3 positive classes\nunique_classes = np.unique(train[class_names].values, axis=0)\nunique_classes # no multi label ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Plot img with bounding box","metadata":{"papermill":{"duration":0.045173,"end_time":"2021-01-01T01:52:00.181219","exception":false,"start_time":"2021-01-01T01:52:00.136046","status":"completed"},"tags":[]}},{"cell_type":"code","source":"imgs = []\nlabel2color = {\n    '[1, 0, 0]': [255,0,0], # Typical Appearance\n    '[0, 1, 0]': [0,255,0], # Indeterminate Appearance\n    '[0, 0, 1]': [0,0,255], # Atypical Appearance\n}\nprint('Typical Appearance: red')\nprint('Indeterminate Appearance: green')\nprint('Atypical Appearance: blue')\nthickness = 3\nscale = 5\n\nfor _, row in train[train['Negative for Pneumonia']==0].iloc[:8].iterrows():\n    study_id = row['StudyInstanceUID']\n    img_path = glob(f'{dataset_dir}/train/{study_id}/*/*')[0]\n    img = dicom2array(path=img_path)\n    img = cv2.resize(img, None, fx=1/scale, fy=1/scale)\n    img = np.stack([img, img, img], axis=-1)\n    \n    claz = row[class_names].values\n    color = label2color[str(claz.tolist())]\n\n    bboxes = []\n    bbox = []\n    for i, l in enumerate(row['label'].split(' ')):\n        if (i % 6 == 0) | (i % 6 == 1):\n            continue\n        bbox.append(float(l)/scale)\n        if i % 6 == 5:\n            bboxes.append(bbox)\n            bbox = []    \n    \n    for box in bboxes:\n        img = cv2.rectangle(\n            img,\n            (int(box[0]), int(box[1])),\n            (int(box[2]), int(box[3])),\n            color, thickness\n    )\n    img = cv2.resize(img, (500,500))\n    imgs.append(img)\n    \nplot_imgs(imgs, cmap=None)","metadata":{"papermill":{"duration":17.072406,"end_time":"2021-01-01T01:52:17.298447","exception":false,"start_time":"2021-01-01T01:52:00.226041","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Typical Appearance only","metadata":{}},{"cell_type":"code","source":"imgs = []\nthickness = 3\nscale = 5\n\nfor _, row in train[train['Typical Appearance'] == 1].iloc[:16].iterrows():\n    study_id = row['StudyInstanceUID']\n    img_path = glob(f'{dataset_dir}/train/{study_id}/*/*')[0]\n    img = dicom2array(path=img_path)\n    img = cv2.resize(img, None, fx=1/scale, fy=1/scale)\n    img = np.stack([img, img, img], axis=-1)\n    \n    claz = row[class_names].values\n    color = label2color[str(claz.tolist())]\n\n    bboxes = []\n    bbox = []\n    for i, l in enumerate(row['label'].split(' ')):\n        if (i % 6 == 0) | (i % 6 == 1):\n            continue\n        bbox.append(float(l)/scale)\n        if i % 6 == 5:\n            bboxes.append(bbox)\n            bbox = []    \n    \n    for box in bboxes:\n        img = cv2.rectangle(\n            img,\n            (int(box[0]), int(box[1])),\n            (int(box[2]), int(box[3])),\n            color, thickness\n    )\n    img = cv2.resize(img, (500,500))\n    imgs.append(img)\n    \nplot_imgs(imgs, cmap=None)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Indeterminate Appearance only","metadata":{}},{"cell_type":"code","source":"imgs = []\nthickness = 3\nscale = 5\n\nfor _, row in train[train['Indeterminate Appearance'] == 1].iloc[:16].iterrows():\n    study_id = row['StudyInstanceUID']\n    img_path = glob(f'{dataset_dir}/train/{study_id}/*/*')[0]\n    img = dicom2array(path=img_path)\n    img = cv2.resize(img, None, fx=1/scale, fy=1/scale)\n    img = np.stack([img, img, img], axis=-1)\n    \n    claz = row[class_names].values\n    color = label2color[str(claz.tolist())]\n\n    bboxes = []\n    bbox = []\n    for i, l in enumerate(row['label'].split(' ')):\n        if (i % 6 == 0) | (i % 6 == 1):\n            continue\n        bbox.append(float(l)/scale)\n        if i % 6 == 5:\n            bboxes.append(bbox)\n            bbox = []    \n    \n    for box in bboxes:\n        img = cv2.rectangle(\n            img,\n            (int(box[0]), int(box[1])),\n            (int(box[2]), int(box[3])),\n            color, thickness\n    )\n    img = cv2.resize(img, (500,500))\n    imgs.append(img)\n    \nplot_imgs(imgs, cmap=None)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Atypical Appearance only","metadata":{}},{"cell_type":"code","source":"imgs = []\nthickness = 3\nscale = 5\n\nfor _, row in train[train['Atypical Appearance'] == 1].iloc[:16].iterrows():\n    study_id = row['StudyInstanceUID']\n    img_path = glob(f'{dataset_dir}/train/{study_id}/*/*')[0]\n    img = dicom2array(path=img_path)\n    img = cv2.resize(img, None, fx=1/scale, fy=1/scale)\n    img = np.stack([img, img, img], axis=-1)\n    \n    claz = row[class_names].values\n    color = label2color[str(claz.tolist())]\n\n    bboxes = []\n    bbox = []\n    \n    for i, l in enumerate(row['label'].split(' ')):\n        if (i % 6 == 0) | (i % 6 == 1):\n            continue\n        bbox.append(float(l)/scale)\n        if i % 6 == 5:\n            bboxes.append(bbox)\n            bbox = []    \n    \n    for box in bboxes:\n        img = cv2.rectangle(\n            img,\n            (int(box[0]), int(box[1])),\n            (int(box[2]), int(box[3])),\n            color, thickness\n    )\n    img = cv2.resize(img, (500,500))\n    imgs.append(img)\n    \nplot_imgs(imgs, cmap=None)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\nsub.loc[sub['id'].str.endswith('study'), 'PredictionString'] = 'negative 1 0 0 1 1 atypical 1 0 0 1 1 typical 1 0 0 1 1 indeterminate 1 0 0 1 1'\nsub.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}