{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-08-04T12:38:01.744609Z","iopub.execute_input":"2021-08-04T12:38:01.74499Z","iopub.status.idle":"2021-08-04T12:38:34.778749Z","shell.execute_reply.started":"2021-08-04T12:38:01.744909Z","shell.execute_reply":"2021-08-04T12:38:34.777742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!conda install gdcm -c conda-forge -y\n","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:38:34.780073Z","iopub.execute_input":"2021-08-04T12:38:34.780359Z","iopub.status.idle":"2021-08-04T12:39:35.630193Z","shell.execute_reply.started":"2021-08-04T12:38:34.780329Z","shell.execute_reply":"2021-08-04T12:39:35.629114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport glob\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport cv2\nimport matplotlib.pyplot as plt\nimport numpy as np \nimport pandas as pd\nfrom tqdm import tqdm\n\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:35.632658Z","iopub.execute_input":"2021-08-04T12:39:35.633082Z","iopub.status.idle":"2021-08-04T12:39:36.184587Z","shell.execute_reply.started":"2021-08-04T12:39:35.633034Z","shell.execute_reply":"2021-08-04T12:39:36.18377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT_PATH = '/kaggle/input/siim-covid19-detection' ","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:36.185784Z","iopub.execute_input":"2021-08-04T12:39:36.186188Z","iopub.status.idle":"2021-08-04T12:39:36.189814Z","shell.execute_reply.started":"2021-08-04T12:39:36.186159Z","shell.execute_reply":"2021-08-04T12:39:36.188895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image_level = pd.read_csv(f'{INPUT_PATH}/train_image_level.csv')","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:36.191045Z","iopub.execute_input":"2021-08-04T12:39:36.19147Z","iopub.status.idle":"2021-08-04T12:39:36.24985Z","shell.execute_reply.started":"2021-08-04T12:39:36.19144Z","shell.execute_reply":"2021-08-04T12:39:36.248732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image_level.head(10)","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:36.251175Z","iopub.execute_input":"2021-08-04T12:39:36.251439Z","iopub.status.idle":"2021-08-04T12:39:36.280514Z","shell.execute_reply.started":"2021-08-04T12:39:36.251412Z","shell.execute_reply":"2021-08-04T12:39:36.279532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Picture-level training set data volume:\", len(train_image_level))","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:36.281648Z","iopub.execute_input":"2021-08-04T12:39:36.281922Z","iopub.status.idle":"2021-08-04T12:39:36.287078Z","shell.execute_reply.started":"2021-08-04T12:39:36.281889Z","shell.execute_reply":"2021-08-04T12:39:36.286153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_type = dict()\nfor x in train_image_level[['label']].iterrows():\n    label = x[1].values[0].split(' ')[0]\n    if label not in label_type:\n        label_type[label] = 0\n    label_type[label] += 1\nprint(label_type)\n","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:36.290274Z","iopub.execute_input":"2021-08-04T12:39:36.290689Z","iopub.status.idle":"2021-08-04T12:39:36.765541Z","shell.execute_reply.started":"2021-08-04T12:39:36.290658Z","shell.execute_reply":"2021-08-04T12:39:36.764381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image_level['boxes'][0]","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:36.767296Z","iopub.execute_input":"2021-08-04T12:39:36.76762Z","iopub.status.idle":"2021-08-04T12:39:36.773987Z","shell.execute_reply.started":"2021-08-04T12:39:36.767583Z","shell.execute_reply":"2021-08-04T12:39:36.77293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image_level['label'][0]","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:36.775712Z","iopub.execute_input":"2021-08-04T12:39:36.776095Z","iopub.status.idle":"2021-08-04T12:39:36.785683Z","shell.execute_reply.started":"2021-08-04T12:39:36.776054Z","shell.execute_reply":"2021-08-04T12:39:36.784868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_study_level = pd.read_csv(f'{INPUT_PATH}/train_study_level.csv')","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:36.786719Z","iopub.execute_input":"2021-08-04T12:39:36.787093Z","iopub.status.idle":"2021-08-04T12:39:36.808348Z","shell.execute_reply.started":"2021-08-04T12:39:36.787059Z","shell.execute_reply":"2021-08-04T12:39:36.807468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_study_level.head(5)","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:36.809595Z","iopub.execute_input":"2021-08-04T12:39:36.809878Z","iopub.status.idle":"2021-08-04T12:39:36.821356Z","shell.execute_reply.started":"2021-08-04T12:39:36.809849Z","shell.execute_reply":"2021-08-04T12:39:36.820212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"The amount of training data at the research level:\", len(train_study_level))","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:36.822913Z","iopub.execute_input":"2021-08-04T12:39:36.823278Z","iopub.status.idle":"2021-08-04T12:39:36.827813Z","shell.execute_reply.started":"2021-08-04T12:39:36.823248Z","shell.execute_reply":"2021-08-04T12:39:36.827028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_study_level.values.tolist()[0][1:5]","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:36.829043Z","iopub.execute_input":"2021-08-04T12:39:36.829296Z","iopub.status.idle":"2021-08-04T12:39:36.841874Z","shell.execute_reply.started":"2021-08-04T12:39:36.829262Z","shell.execute_reply":"2021-08-04T12:39:36.841074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_study_level1 = train_study_level.copy(deep=True)\ntrain_study_level1['StudyInstanceUID'] = train_study_level1['id'].apply(lambda x: x.replace('_study', ''))\n\n# 检查有没有重复的study号\nflag = train_study_level1['StudyInstanceUID'].duplicated()\nprint(\"Whether there is a duplicate study number：\", flag.any())\n\n# 删除id并进行image水平和study水平融合\ndel train_study_level1['id']\ntrain_df = pd.merge(train_image_level, train_study_level1, how='left', on='StudyInstanceUID')\n\ngroup_col = 'StudyInstanceUID'\ncount_df=pd.DataFrame(train_df.groupby(group_col)['id'].count())\ncount_df.columns = [f'{group_col}_count']\ntrain_df=train_df.merge(count_df.reset_index(), on=group_col)\ntrain_df.head(2)\n\none_study_multi_image_df = train_df[train_df[f'{group_col}_count'] > 1]\none_study_multi_image_df.head(5)\nprint(\"Number of special cases：\", len(one_study_multi_image_df))\n\n# 删除特殊情况\ntrain_df = train_df[train_df[f'{group_col}_count'] == 1] # delete 'StudyInstanceUID_count > 1' data\nprint(\"Sample size after removing special cases：\", len(train_df))","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:36.84311Z","iopub.execute_input":"2021-08-04T12:39:36.843438Z","iopub.status.idle":"2021-08-04T12:39:36.906473Z","shell.execute_reply.started":"2021-08-04T12:39:36.843406Z","shell.execute_reply":"2021-08-04T12:39:36.905801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv(f'{INPUT_PATH}/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:36.907746Z","iopub.execute_input":"2021-08-04T12:39:36.908101Z","iopub.status.idle":"2021-08-04T12:39:36.922316Z","shell.execute_reply.started":"2021-08-04T12:39:36.90807Z","shell.execute_reply":"2021-08-04T12:39:36.921593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:36.923307Z","iopub.execute_input":"2021-08-04T12:39:36.923705Z","iopub.status.idle":"2021-08-04T12:39:36.935052Z","shell.execute_reply.started":"2021-08-04T12:39:36.923674Z","shell.execute_reply":"2021-08-04T12:39:36.933873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission.info()","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:48:52.000677Z","iopub.execute_input":"2021-08-04T12:48:52.002103Z","iopub.status.idle":"2021-08-04T12:48:52.023438Z","shell.execute_reply.started":"2021-08-04T12:48:52.002038Z","shell.execute_reply":"2021-08-04T12:48:52.022446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"train:\",len(os.listdir(f'{INPUT_PATH}/train')), len(glob.glob(f'{INPUT_PATH}/train/*/*/*.dcm')))\nprint(\"test :\",len(os.listdir(f'{INPUT_PATH}/test')), len(glob.glob(f'{INPUT_PATH}/test/*/*/*.dcm')))","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:36.936421Z","iopub.execute_input":"2021-08-04T12:39:36.937023Z","iopub.status.idle":"2021-08-04T12:39:43.998049Z","shell.execute_reply.started":"2021-08-04T12:39:36.936979Z","shell.execute_reply":"2021-08-04T12:39:43.997084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def dicom2array(path, voi_lut=True, fix_monochrome=True):\n    dicom = pydicom.read_file(path)\n    # VOI LUT (if available by DICOM device) is used to\n    # transform raw DICOM data to \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n        \n    \ndef plot_img(img, size=(7, 7), is_rgb=True, title=\"\", cmap='gray'):\n    plt.figure(figsize=size)\n    plt.imshow(img, cmap=cmap)\n    plt.suptitle(title)\n    plt.show()\n\ndef plot_imgs(imgs, cols=4, size=7, is_rgb=True, title=\"\", cmap='gray', img_size=(500,500)):\n    rows = len(imgs)//cols + 1\n    fig = plt.figure(figsize=(cols*size, rows*size))\n    for i, img in enumerate(imgs):\n        if img_size is not None:\n            img = cv2.resize(img, img_size)\n        fig.add_subplot(rows, cols, i+1)\n        plt.imshow(img, cmap=cmap)\n    plt.suptitle(title)\n    plt.show()\n\ndicom_paths = glob.glob(f'{INPUT_PATH}/train/*/*/*.dcm')\nimgs = [dicom2array(path) for path in dicom_paths[:4]]\nprint(imgs[0].shape, imgs[1].shape)  # 注意，(2336, 2836) (3488, 4256) 图片大小不一致\nplot_imgs(imgs)\n","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:43.999316Z","iopub.execute_input":"2021-08-04T12:39:43.99964Z","iopub.status.idle":"2021-08-04T12:39:52.79469Z","shell.execute_reply.started":"2021-08-04T12:39:43.999606Z","shell.execute_reply":"2021-08-04T12:39:52.793589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pie_plot(train_df, variable):\n    \"\"\"\n        input: variable ex: \"Sex\"\n        output: bar plot & value count\n    \"\"\"\n    # get feature\n    var = train_df[variable]\n    # count number of categorical variable(value/sample)\n    varValue = var.value_counts()\n    \n    # visualize\n    plt.figure(figsize = (20,10))\n    plt.pie(varValue, labels=varValue.index, autopct=\"%1.1f%%\")\n#     plt.xticks(varValue.index, varValue.index.values)\n#     plt.ylabel(\"Frequency\")\n    plt.title('target')\n    plt.show()\n    \ntrain_df['target'] = 'Negative for Pneumonia'\ntrain_df.loc[train_df['Typical Appearance']==1, 'target'] = 'Typical Appearance'\ntrain_df.loc[train_df['Indeterminate Appearance']==1, 'target'] = 'Indeterminate Appearance'\ntrain_df.loc[train_df['Atypical Appearance']==1, 'target'] = 'Atypical Appearance'\nprint(train_df['target'].value_counts())\npie_plot(train_df, 'target')   ","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:52.79626Z","iopub.execute_input":"2021-08-04T12:39:52.796678Z","iopub.status.idle":"2021-08-04T12:39:52.941857Z","shell.execute_reply.started":"2021-08-04T12:39:52.796634Z","shell.execute_reply":"2021-08-04T12:39:52.94064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['boxes'].values[0] # x_min, y_min, width, height","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:52.942953Z","iopub.execute_input":"2021-08-04T12:39:52.943218Z","iopub.status.idle":"2021-08-04T12:39:52.949356Z","shell.execute_reply.started":"2021-08-04T12:39:52.943191Z","shell.execute_reply":"2021-08-04T12:39:52.948385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['label'].values[0] # x_min, y_min, x_max, y_max","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:52.950911Z","iopub.execute_input":"2021-08-04T12:39:52.951252Z","iopub.status.idle":"2021-08-04T12:39:52.967064Z","shell.execute_reply.started":"2021-08-04T12:39:52.951221Z","shell.execute_reply":"2021-08-04T12:39:52.965947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_names = ['Typical Appearance', 'Indeterminate Appearance', 'Atypical Appearance'] # we have 3 positive classes\nunique_classes = np.unique(train_df[class_names].values, axis=0)  \nunique_classes","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:52.970261Z","iopub.execute_input":"2021-08-04T12:39:52.970594Z","iopub.status.idle":"2021-08-04T12:39:52.99036Z","shell.execute_reply.started":"2021-08-04T12:39:52.970534Z","shell.execute_reply":"2021-08-04T12:39:52.98901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label2color = {\n    '[1, 0, 0]': [255,0,0], # Typical Appearance\n    '[0, 1, 0]': [0,255,0], # Indeterminate Appearance\n    '[0, 0, 1]': [0,0,255], # Atypical Appearance\n    '[0, 0, 0]': None, # negative\n}\n\nlabel2target = {\n    '[1, 0, 0]': 'typical',\n    '[0, 1, 0]': 'indeterminate',\n    '[0, 0, 1]': 'atypical',\n    '[0, 0, 0]': 'negative'\n}","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:52.992004Z","iopub.execute_input":"2021-08-04T12:39:52.992273Z","iopub.status.idle":"2021-08-04T12:39:52.997797Z","shell.execute_reply.started":"2021-08-04T12:39:52.992246Z","shell.execute_reply":"2021-08-04T12:39:52.99669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"THICKNESS = 3\nSCALE = 5\nFONT = cv2.FONT_HERSHEY_SIMPLEX; FONT_SCALE = 1; FONT_THICKNESS = 2; FONT_LINE_TYPE = cv2.LINE_AA;","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:52.998943Z","iopub.execute_input":"2021-08-04T12:39:52.999305Z","iopub.status.idle":"2021-08-04T12:39:53.011539Z","shell.execute_reply.started":"2021-08-04T12:39:52.999275Z","shell.execute_reply":"2021-08-04T12:39:53.01033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot img with bounding box\nimgs = []\nfor _, row in train_df.iloc[:8].iterrows():\n    study_id = row['StudyInstanceUID']\n    img_path = glob.glob(f'{INPUT_PATH}/train/{study_id}/*/*.dcm')[0]\n    img = dicom2array(path=img_path)\n    img = cv2.resize(img, None, fx=1/SCALE, fy=1/SCALE) # 尺度变换\n    img = np.stack([img, img, img], axis=-1)  # 灰度图像转RGB 堆叠\n    \n    claz = row[class_names].values\n    color = label2color[str(claz.tolist())]\n    target = label2target[str(claz.tolist())]\n\n    bboxes = []\n    bbox = []\n    for i, l in enumerate(row['label'].split(' ')): \n        # 'opacity 1 789.28836 582.43035 1815.94498 2499.73327 opacity 1 2245.91208 591.20528 3340.5737 2352.75472'\n        if (i % 6 == 0) | (i % 6 == 1):\n            continue\n        bbox.append(float(l)/SCALE)\n        if i % 6 == 5:\n            bboxes.append(bbox)\n            bbox = []    \n\n    text_width, text_height = cv2.getTextSize(target, FONT, FONT_SCALE, FONT_THICKNESS)[0]\n    for box in bboxes:\n        img = cv2.rectangle(\n            img,\n            (int(box[0]), int(box[1])),\n            (int(box[2]), int(box[3])),\n            color, THICKNESS\n        )\n        box_width = int(box[2]) - int(box[0])\n        img = cv2.putText(img, target, (int(box[0])-(text_width-box_width)//2, int(box[1])-10),\n                        FONT, FONT_SCALE, color, FONT_THICKNESS, FONT_LINE_TYPE)          \n    img = cv2.resize(img, (500, 500))\n    imgs.append(img)\n    \nplot_imgs(imgs, cmap=None)\n","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:53.013171Z","iopub.execute_input":"2021-08-04T12:39:53.013605Z","iopub.status.idle":"2021-08-04T12:39:56.998434Z","shell.execute_reply.started":"2021-08-04T12:39:53.013542Z","shell.execute_reply":"2021-08-04T12:39:56.997467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs = []\nfor _, row in train_df[train_df['Typical Appearance'] == 1].iloc[:8].iterrows():\n    study_id = row['StudyInstanceUID']\n    img_path = glob.glob(f'{INPUT_PATH}/train/{study_id}/*/*.dcm')[0]\n    img = dicom2array(path=img_path)\n    img = cv2.resize(img, None, fx=1/SCALE, fy=1/SCALE)\n    img = np.stack([img, img, img], axis=-1)\n    \n    claz = row[class_names].values\n    color = label2color[str(claz.tolist())]\n    target = label2target[str(claz.tolist())]\n\n    bboxes = []\n    bbox = []\n    for i, l in enumerate(row['label'].split(' ')):\n        if (i % 6 == 0) | (i % 6 == 1):\n            continue\n        bbox.append(float(l)/SCALE)\n        if i % 6 == 5:\n            bboxes.append(bbox)\n            bbox = []    \n    text_width, text_height = cv2.getTextSize(target, FONT, FONT_SCALE, FONT_THICKNESS)[0]\n    for box in bboxes:\n        img = cv2.rectangle(\n            img,\n            (int(box[0]), int(box[1])),\n            (int(box[2]), int(box[3])),\n            color, THICKNESS\n    \t)\n        box_width = int(box[2]) - int(box[0])\n        img = cv2.putText(img, target, (int(box[0])-(text_width-box_width)//2, int(box[1])-10),\n                        FONT, FONT_SCALE, color, FONT_THICKNESS, FONT_LINE_TYPE) \n    img = cv2.resize(img, (500,500))\n    imgs.append(img)\n    \nplot_imgs(imgs, cmap=None)\n","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:39:56.999645Z","iopub.execute_input":"2021-08-04T12:39:56.999917Z","iopub.status.idle":"2021-08-04T12:40:00.144646Z","shell.execute_reply.started":"2021-08-04T12:39:56.999888Z","shell.execute_reply":"2021-08-04T12:40:00.143619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs = []\nfor _, row in train_df[train_df['Indeterminate Appearance'] == 1].iloc[:8].iterrows():\n    study_id = row['StudyInstanceUID']\n    img_path = glob.glob(f'{INPUT_PATH}/train/{study_id}/*/*.dcm')[0]\n    img = dicom2array(path=img_path)\n    img = cv2.resize(img, None, fx=1/SCALE, fy=1/SCALE)\n    img = np.stack([img, img, img], axis=-1)\n    \n    claz = row[class_names].values\n    color = label2color[str(claz.tolist())]\n    target = label2target[str(claz.tolist())]\n\n    bboxes = []\n    bbox = []\n    for i, l in enumerate(row['label'].split(' ')):\n        if (i % 6 == 0) | (i % 6 == 1):\n            continue\n        bbox.append(float(l)/SCALE)\n        if i % 6 == 5:\n            bboxes.append(bbox)\n            bbox = []    \n    \n    for box in bboxes:\n        img = cv2.rectangle(\n            img,\n            (int(box[0]), int(box[1])),\n            (int(box[2]), int(box[3])),\n            color, THICKNESS\n    \t)\n        box_width = int(box[2]) - int(box[0])\n        img = cv2.putText(img, target, (int(box[0])-(text_width-box_width)//2, int(box[1])-10),\n                        FONT, FONT_SCALE, color, FONT_THICKNESS, FONT_LINE_TYPE) \n    img = cv2.resize(img, (500,500))\n    imgs.append(img)\n    \nplot_imgs(imgs, cmap=None)\n","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:40:00.14602Z","iopub.execute_input":"2021-08-04T12:40:00.146509Z","iopub.status.idle":"2021-08-04T12:40:04.21699Z","shell.execute_reply.started":"2021-08-04T12:40:00.146469Z","shell.execute_reply":"2021-08-04T12:40:04.213403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs = []\nfor _, row in train_df[train_df['Atypical Appearance'] == 1].iloc[:8].iterrows():\n    study_id = row['StudyInstanceUID']\n    img_path = glob.glob(f'{INPUT_PATH}/train/{study_id}/*/*.dcm')[0]\n    img = dicom2array(path=img_path)\n    img = cv2.resize(img, None, fx=1/SCALE, fy=1/SCALE)\n    img = np.stack([img, img, img], axis=-1)\n    \n    claz = row[class_names].values\n    color = label2color[str(claz.tolist())]\n    target = label2target[str(claz.tolist())]\n\n    bboxes = []\n    bbox = []\n    for i, l in enumerate(row['label'].split(' ')):\n        if (i % 6 == 0) | (i % 6 == 1):\n            continue\n        bbox.append(float(l)/SCALE)\n        if i % 6 == 5:\n            bboxes.append(bbox)\n            bbox = []    \n    \n    for box in bboxes:\n        img = cv2.rectangle(\n            img,\n            (int(box[0]), int(box[1])),\n            (int(box[2]), int(box[3])),\n            color, THICKNESS\n    \t)\n        box_width = int(box[2]) - int(box[0])\n        img = cv2.putText(img, target, (int(box[0])-(text_width-box_width)//2, int(box[1])-10),\n                        FONT, FONT_SCALE, color, FONT_THICKNESS, FONT_LINE_TYPE) \n    img = cv2.resize(img, (500,500))\n    imgs.append(img)\n    \nplot_imgs(imgs, cmap=None)\n","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:40:04.218099Z","iopub.execute_input":"2021-08-04T12:40:04.218506Z","iopub.status.idle":"2021-08-04T12:40:08.307351Z","shell.execute_reply.started":"2021-08-04T12:40:04.218469Z","shell.execute_reply":"2021-08-04T12:40:08.306662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs = []\n\nfor i in range(8):\n    if i % 2 == 0:\n        row = train_df[train_df['Negative for Pneumonia']==1].iloc[i]\n    else:\n        row = train_df[train_df['Typical Appearance']==1].iloc[i]\n        \n    study_id = row['StudyInstanceUID']\n    img_path = glob.glob(f'{INPUT_PATH}/train/{study_id}/*/*')[0]\n    img = dicom2array(path=img_path)\n    img = cv2.resize(img, None, fx=1/SCALE, fy=1/SCALE)\n    img = np.stack([img, img, img], axis=-1)\n    \n    claz = row[class_names].values\n    color = label2color[str(claz.tolist())]\n    target = label2target[str(claz.tolist())]\n\n    bboxes = []\n    bbox = []\n    for i, l in enumerate(row['label'].split(' ')):\n        if (i % 6 == 0) | (i % 6 == 1):\n            continue\n        bbox.append(float(l)/SCALE)\n        if i % 6 == 5:\n            bboxes.append(bbox)\n            bbox = []    \n    \n    for box in bboxes:\n        img = cv2.rectangle(\n            img,\n            (int(box[0]), int(box[1])),\n            (int(box[2]), int(box[3])),\n            color, THICKNESS\n    \t)\n        box_width = int(box[2]) - int(box[0])\n        img = cv2.putText(img, target, (int(box[0])-(text_width-box_width)//2, int(box[1])-10),\n                        FONT, FONT_SCALE, color, FONT_THICKNESS, FONT_LINE_TYPE) \n    img = cv2.resize(img, (500,500))\n    imgs.append(img)\n    \nplot_imgs(imgs, cmap=None)\n","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:40:08.308412Z","iopub.execute_input":"2021-08-04T12:40:08.308816Z","iopub.status.idle":"2021-08-04T12:40:10.869036Z","shell.execute_reply.started":"2021-08-04T12:40:08.30878Z","shell.execute_reply":"2021-08-04T12:40:10.867836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2021-08-04T13:01:05.313151Z","iopub.execute_input":"2021-08-04T13:01:05.313498Z","iopub.status.idle":"2021-08-04T13:01:05.323894Z","shell.execute_reply.started":"2021-08-04T13:01:05.313466Z","shell.execute_reply":"2021-08-04T13:01:05.322890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}