{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        # print(os.path.join(dirname, filename))\n        pass\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-05-28T07:42:32.032906Z","iopub.execute_input":"2021-05-28T07:42:32.033298Z","iopub.status.idle":"2021-05-28T07:43:00.405867Z","shell.execute_reply.started":"2021-05-28T07:42:32.033220Z","shell.execute_reply":"2021-05-28T07:43:00.404774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nimport numpy as np \nimport random\nimport pandas as pd \nimport missingno as msno\nfrom collections import Counter\nimport glob\nfrom tqdm.notebook import tqdm\nfrom pathlib import Path\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\n\nfrom plotly import graph_objs as go\nimport plotly.express as px\nimport plotly.figure_factory as ff\nfrom plotly.subplots import make_subplots\n\nimport os\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport cv2\n\n# from skimage import exposure","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-05-28T08:06:30.777757Z","iopub.execute_input":"2021-05-28T08:06:30.778275Z","iopub.status.idle":"2021-05-28T08:06:30.792319Z","shell.execute_reply.started":"2021-05-28T08:06:30.778238Z","shell.execute_reply":"2021-05-28T08:06:30.791029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed=0):\n    random.seed(seed)\n    np.random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    os.environ['TF_DETERMINISTIC_OPS'] = '1'\n\nseed = 2021\nseed_everything(seed)\n\nwarnings.filterwarnings('ignore')\npd.set_option('display.max_colwidth', 150)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-05-28T07:43:04.231413Z","iopub.execute_input":"2021-05-28T07:43:04.232018Z","iopub.status.idle":"2021-05-28T07:43:04.239129Z","shell.execute_reply.started":"2021-05-28T07:43:04.231969Z","shell.execute_reply":"2021-05-28T07:43:04.238465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defining all our palette colours.\nprimary_blue = \"#496595\"\nprimary_blue2 = \"#85a1c1\"\nprimary_blue3 = \"#3f4d63\"\nprimary_grey = \"#c6ccd8\"\nprimary_black = \"#202022\"\nprimary_bgcolor = \"#f4f0ea\"\n\nprimary_green = px.colors.qualitative.Plotly[2]\n\nplotly_discrete_sequence = px.colors.qualitative.G10\n\nplt.rcParams['figure.dpi'] = 120\nplt.rcParams['axes.spines.top'] = False\nplt.rcParams['axes.spines.right'] = False\nplt.rcParams['font.family'] = 'serif'\nplt.rcParams['axes.facecolor'] = primary_bgcolor\n\ncolors = [primary_blue, primary_blue2, primary_blue3, primary_grey, primary_black, primary_bgcolor, primary_green]\nsns.palplot(sns.color_palette(colors))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-05-28T07:43:04.240487Z","iopub.execute_input":"2021-05-28T07:43:04.240877Z","iopub.status.idle":"2021-05-28T07:43:04.347476Z","shell.execute_reply.started":"2021-05-28T07:43:04.240851Z","shell.execute_reply":"2021-05-28T07:43:04.346756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.palplot(sns.color_palette(plotly_discrete_sequence))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-05-28T07:43:04.348478Z","iopub.execute_input":"2021-05-28T07:43:04.348845Z","iopub.status.idle":"2021-05-28T07:43:04.440561Z","shell.execute_reply.started":"2021-05-28T07:43:04.348817Z","shell.execute_reply":"2021-05-28T07:43:04.439640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <p style=\"background-color:skyblue; font-family:newtimeroman; font-size:250%; text-align:center; border-radius: 15px 50px;\">🦠SIIM Covid-19🦟 EDA and Visualization 📊</p>\n\nThis is an **object detection** and **classification problem**, meaning that for each instance we'll have to predict a bounding box and a class. It seems to be a multi-label problem cuz there are 4 columns per image, but as they are auto-axclusive, the challenge is a multi-class problem.\n\n <div class=\"alert alert-success\" role=\"alert\">\n    <p>💡 <b>Competition Goal</b>: Categorize chest radiographs as negative for pneumonia, typical, indeterminate, or atypical for COVID-19. If some abnormalities are found, provide the bounding boxes. </p>\n</div>\n\nWe can see that we have:\n* `train_study_level.csv` - the train study-level metadata, with one row for each study, including correct labels.\n* `train_image_level.csv` - the train image-level metadata, with one row for each image, including both correct labels and any bounding boxes in a dictionary format. Some images in both test and train have multiple bounding boxes.\n* `sample_submission.csv` - a sample submission file containing all image- and study-level IDs.\n* train folder - comprises chest scans in DICOM format, stored in paths following the schema: `study/series/image`\n* test folder - The hidden test dataset is of roughly the same scale as the training dataset.","metadata":{}},{"cell_type":"code","source":"virus_url = 'https://upload.wikimedia.org/wikipedia/commons/2/21/Virus_gray_black.svg'","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-05-28T07:43:04.441802Z","iopub.execute_input":"2021-05-28T07:43:04.442090Z","iopub.status.idle":"2021-05-28T07:43:04.446643Z","shell.execute_reply.started":"2021-05-28T07:43:04.442062Z","shell.execute_reply":"2021-05-28T07:43:04.445638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read in metadata\ntrain_study_df = pd.read_csv(\"../input/siim-covid19-detection/train_study_level.csv\")\ntrain_image_df = pd.read_csv(\"../input/siim-covid19-detection/train_image_level.csv\")\n\nprint(f\"Train Study Shape: {train_study_df.shape} \\n\" +\n      f\"Train Image Shape: {train_image_df.shape} \\n\" + \"\\n\" +\n      f\"Note: There are {train_image_df['boxes'].isna().sum()} missing values in train_image.\")","metadata":{"execution":{"iopub.status.busy":"2021-05-28T07:43:04.447854Z","iopub.execute_input":"2021-05-28T07:43:04.448106Z","iopub.status.idle":"2021-05-28T07:43:04.524779Z","shell.execute_reply.started":"2021-05-28T07:43:04.448081Z","shell.execute_reply":"2021-05-28T07:43:04.523736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_study_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-28T07:43:04.526755Z","iopub.execute_input":"2021-05-28T07:43:04.527089Z","iopub.status.idle":"2021-05-28T07:43:04.544431Z","shell.execute_reply.started":"2021-05-28T07:43:04.527057Z","shell.execute_reply":"2021-05-28T07:43:04.543568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-28T07:43:04.545802Z","iopub.execute_input":"2021-05-28T07:43:04.546248Z","iopub.status.idle":"2021-05-28T07:43:04.558089Z","shell.execute_reply.started":"2021-05-28T07:43:04.546215Z","shell.execute_reply":"2021-05-28T07:43:04.556975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id='1'></a>\n# <p style=\"background-color:skyblue; font-family:newtimeroman; font-size:150%; text-align:center; border-radius: 15px 50px;\">1. First overview and EDA 📊</p>\n\nThe bounding box labels are provided in the `label` column. The format is as follows:\n\n`[class ID] [confidence score] [bounding box]`\n\n`class ID` - either opacity or none\n`confidence score` - confidence from your neural network model. If none, the confidence is 1.\n`bounding box` - typical x0 y0 x1 y1 format. If class ID is none, the bounding box is 1 0 0 1 1.\nThe bounding boxes are also provided in easily readable dictionary format in column boxes, and the study that each image is a part of is provided `inStudyInstanceUID`.\n\nLets take a look about class distribution.","metadata":{}},{"cell_type":"code","source":"train_image_df['class'] = train_image_df['label'].apply(lambda x: x.split(' ')[0])","metadata":{"execution":{"iopub.status.busy":"2021-05-28T07:43:04.559393Z","iopub.execute_input":"2021-05-28T07:43:04.559816Z","iopub.status.idle":"2021-05-28T07:43:04.576325Z","shell.execute_reply.started":"2021-05-28T07:43:04.559776Z","shell.execute_reply":"2021-05-28T07:43:04.575245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_df = train_image_df['class'].value_counts().reset_index()\n\nfig = go.Figure(go.Bar(\n    x = plot_df['class'],\n    y = plot_df['index'],\n    orientation='h',\n    marker_color=[primary_blue, primary_grey],\n    marker_line_color=primary_black,\n    marker_line_width=1.5, \n    opacity=0.8,\n))\n# Change the bar mode\nfig.update_layout(\n    title='<span style=\"font-size:32px; font-family:Serif\"><b>Class sidtribution</b></span>',\n    yaxis_title=f'<b>Class</b>',\n    xaxis_title=f'<b>Count</b>',\n    legend_title=\"Group\",\n    font=dict(\n        family=\"Times New Roman\",\n        size=14,\n    )\n)\nfig.add_layout_image(\n    dict(\n        source=virus_url,\n        xref=\"x\", yref=\"paper\",\n        x=4000, y=0.1,\n        sizex=400, sizey=0.25, \n        xanchor=\"center\", yanchor=\"bottom\",\n        sizing='stretch',\n    ),\n)\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-05-28T07:43:04.577553Z","iopub.execute_input":"2021-05-28T07:43:04.577859Z","iopub.status.idle":"2021-05-28T07:43:04.739467Z","shell.execute_reply.started":"2021-05-28T07:43:04.577832Z","shell.execute_reply":"2021-05-28T07:43:04.738507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's now analyze the `train_study_df` labels.","metadata":{}},{"cell_type":"code","source":"labels = train_study_df.drop(columns='id').columns\n\nfig = make_subplots(\n    rows=2, cols=2,\n    subplot_titles=labels\n)\n\nfor i, label in enumerate(labels):\n    plot_df = train_study_df[label].value_counts().reset_index()\n    \n    fig.add_trace(go.Bar(\n        x = plot_df[label],\n        y = plot_df['index'],\n        orientation='h',\n        marker_color=[primary_grey, primary_blue],\n        marker_line_color=primary_black,\n        marker_line_width=1.5, \n        opacity=0.8,\n        name=label\n    ), row=i//2 + 1, col=i%2 + 1,)\n    \n    fig.add_layout_image(\n        dict(\n            source=virus_url,\n            xref=\"x\", yref=\"y domain\",\n            x=plot_df[plot_df['index'] == 1][label].values[0] * 0.7, y=0.6,\n            sizex=320, sizey=0.24, \n            xanchor=\"center\", yanchor=\"bottom\",\n            sizing='stretch',\n        ), row=i//2 + 1, col=i%2 + 1,\n    )\n    \n# Change the bar mode\nfig.update_layout(\n    title='<span style=\"font-size:32px; font-family:Serif\"><b>Label sidtribution</b> in study data</span>',\n    showlegend=False,\n)\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-05-28T07:43:04.741040Z","iopub.execute_input":"2021-05-28T07:43:04.741448Z","iopub.status.idle":"2021-05-28T07:43:04.991504Z","shell.execute_reply.started":"2021-05-28T07:43:04.741404Z","shell.execute_reply":"2021-05-28T07:43:04.990487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id='2'></a>\n# <p style=\"background-color:skyblue; font-family:newtimeroman; font-size:150%; text-align:center; border-radius: 15px 50px;\">2. Lets visaulize the images</p>","metadata":{}},{"cell_type":"code","source":"dataset_path = Path('../input/siim-covid19-detection')\nlist(dataset_path.iterdir())","metadata":{"execution":{"iopub.status.busy":"2021-05-28T08:10:46.459335Z","iopub.execute_input":"2021-05-28T08:10:46.459891Z","iopub.status.idle":"2021-05-28T08:10:46.469941Z","shell.execute_reply.started":"2021-05-28T08:10:46.459842Z","shell.execute_reply":"2021-05-28T08:10:46.468789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To be continue...","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}