{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In this notebook I do simple EDA, ending with plotting some of the training data","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-27T08:29:50.278528Z","iopub.execute_input":"2021-06-27T08:29:50.279361Z","iopub.status.idle":"2021-06-27T08:29:50.289571Z","shell.execute_reply.started":"2021-06-27T08:29:50.279238Z","shell.execute_reply":"2021-06-27T08:29:50.288741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom matplotlib import patches, patheffects\nimport imageio","metadata":{"execution":{"iopub.status.busy":"2021-06-27T08:29:50.312277Z","iopub.execute_input":"2021-06-27T08:29:50.312648Z","iopub.status.idle":"2021-06-27T08:29:50.368730Z","shell.execute_reply.started":"2021-06-27T08:29:50.312615Z","shell.execute_reply":"2021-06-27T08:29:50.367578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_image_locs(dirname='../input/siim-covid19-detection/train'):\n    collect = []\n    for dirname, _, filenames in os.walk(dirname):\n        for filename in filenames:\n            study, series = dirname.split(\"/\")[-2:]\n            image = filename.replace(\".dcm\", \"\")\n            collect.append((study, series, image))\n    locs = pd.DataFrame(collect, columns=(\"study\", \"series\", \"image\"))\n    return locs\n\ntrain_image_locs = read_image_locs()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T08:29:50.370687Z","iopub.execute_input":"2021-06-27T08:29:50.371056Z","iopub.status.idle":"2021-06-27T08:30:20.386770Z","shell.execute_reply.started":"2021-06-27T08:29:50.371023Z","shell.execute_reply":"2021-06-27T08:30:20.385441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_level = pd.read_csv(\"../input/siim-covid19-detection/train_image_level.csv\")\nstudy_level = pd.read_csv(\"../input/siim-covid19-detection/train_study_level.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-06-27T08:30:20.389328Z","iopub.execute_input":"2021-06-27T08:30:20.389754Z","iopub.status.idle":"2021-06-27T08:30:20.465683Z","shell.execute_reply.started":"2021-06-27T08:30:20.389708Z","shell.execute_reply":"2021-06-27T08:30:20.464612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_level.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T08:30:20.467702Z","iopub.execute_input":"2021-06-27T08:30:20.468076Z","iopub.status.idle":"2021-06-27T08:30:20.496992Z","shell.execute_reply.started":"2021-06-27T08:30:20.468042Z","shell.execute_reply":"2021-06-27T08:30:20.495962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_level.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T08:30:20.498408Z","iopub.execute_input":"2021-06-27T08:30:20.498719Z","iopub.status.idle":"2021-06-27T08:30:20.513901Z","shell.execute_reply.started":"2021-06-27T08:30:20.498686Z","shell.execute_reply":"2021-06-27T08:30:20.512967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here I merge the study and image dataframes to get study id and labels associated with each image","metadata":{}},{"cell_type":"code","source":"study_level_ = study_level.copy()\nstudy_level_.id = study_level_.id.str.replace(\"_study\", \"\")\nmerged = image_level.merge(study_level_, left_on=\"StudyInstanceUID\", right_on=\"id\", suffixes=(\"_image\", \"_study\"))\nmerged = merged.rename({'Negative for Pneumonia': 'Negative', 'Typical Appearance': 'Typical', 'Indeterminate Appearance': 'Indeterminate', 'Atypical Appearance': 'Atypical'}, \n              axis=\"columns\")\nmerged.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T08:30:20.515604Z","iopub.execute_input":"2021-06-27T08:30:20.516160Z","iopub.status.idle":"2021-06-27T08:30:20.563297Z","shell.execute_reply.started":"2021-06-27T08:30:20.516106Z","shell.execute_reply":"2021-06-27T08:30:20.562296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here I check that the merged dataframe contains the same number of rows as the image dataframe","metadata":{}},{"cell_type":"code","source":"merged.shape, image_level.shape, study_level.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-27T08:30:20.575386Z","iopub.execute_input":"2021-06-27T08:30:20.575785Z","iopub.status.idle":"2021-06-27T08:30:20.586538Z","shell.execute_reply.started":"2021-06-27T08:30:20.575751Z","shell.execute_reply":"2021-06-27T08:30:20.585196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here I check that the last four columns are indeed one hot encoded, i.e. that exactly one of these equals one, while the three equals zero for each row","metadata":{}},{"cell_type":"code","source":"(merged[['Negative', 'Typical', 'Indeterminate', 'Atypical']].sum(axis=\"columns\") != 1).sum()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T08:36:45.294488Z","iopub.execute_input":"2021-06-27T08:36:45.295107Z","iopub.status.idle":"2021-06-27T08:36:45.323160Z","shell.execute_reply.started":"2021-06-27T08:36:45.295042Z","shell.execute_reply":"2021-06-27T08:36:45.321953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Plot X-ray","metadata":{}},{"cell_type":"code","source":"def get_boxes_and_labels(image_info):\n    if pd.isna(image_info.boxes):\n        boxes, labels = [], []\n    else:\n        boxes = eval(image_info.boxes)\n        label = image_info[['Negative', 'Typical', 'Indeterminate', 'Atypical']].T\n        label = label[label.astype(\"bool\")].index[0]\n        labels = len(boxes)*[label]\n    return boxes, labels\n    \ndef draw_boxes(ax, image_info, color=\"tab:orange\"):\n    for box, label in zip(*get_boxes_and_labels(image_info)):\n        rect = patches.Rectangle((box[\"x\"], box[\"y\"]), box[\"width\"], box[\"width\"], fill=False, edgecolor=color)\n        ax.add_patch(rect)\n        text = ax.text(box[\"x\"], box[\"y\"], label, color=color, fontsize=20)\n        text.set_path_effects([patheffects.Stroke(linewidth=3, foreground='black'),\n                   patheffects.Normal()]) \ndef plot_xray(idx, ax):\n    image_info = merged.iloc[idx]\n    image_loc = train_image_locs[train_image_locs[\"study\"] == image_info[\"StudyInstanceUID\"]]\n    image_file = '../input/siim-covid19-detection/train/' + \"/\".join(image_loc.values[0,:]) + \".dcm\"\n    im = imageio.imread(image_file)\n    ax.imshow(im, cmap=\"gray\")\n    ax.axis(\"off\")\n    \n    draw_boxes(ax, image_info)","metadata":{"execution":{"iopub.status.busy":"2021-06-27T08:33:20.538819Z","iopub.execute_input":"2021-06-27T08:33:20.539260Z","iopub.status.idle":"2021-06-27T08:33:20.551619Z","shell.execute_reply.started":"2021-06-27T08:33:20.539217Z","shell.execute_reply":"2021-06-27T08:33:20.550614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(4, 4, figsize=(20, 20))\nidxs = [i for i in range(axes.size)]\nprint(type(axes))\nfor idx, ax in zip(idxs, axes.flat):\n    plot_xray(idx, ax)","metadata":{"execution":{"iopub.status.busy":"2021-06-27T08:33:20.924878Z","iopub.execute_input":"2021-06-27T08:33:20.925302Z","iopub.status.idle":"2021-06-27T08:33:36.427490Z","shell.execute_reply.started":"2021-06-27T08:33:20.925262Z","shell.execute_reply":"2021-06-27T08:33:36.425793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The brightness varies a lot between each image, so some form of brightness equilization should be done.","metadata":{}}]}