{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<img src = \"https://i.imgur.com/LKLpFOv.png\">","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nfrom tqdm import tqdm\nimport glob\nimport os\nimport matplotlib.pyplot as plt\nimport matplotlib.pylab as pylab\nimport seaborn as sns\nimport pprint\nimport pydicom as dicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport albumentations as A\nimport cv2\nimport wandb\n\nfrom PIL import Image\nfrom colorama import Fore, Back, Style\n# colored output\ny_ = Fore.YELLOW\nr_ = Fore.RED\ng_ = Fore.GREEN\nb_ = Fore.BLUE\nm_ = Fore.MAGENTA\n\nsns.set(font=\"Serif\",style =\"white\")","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-05-22T22:35:50.333638Z","iopub.execute_input":"2021-05-22T22:35:50.333967Z","iopub.status.idle":"2021-05-22T22:35:50.34029Z","shell.execute_reply.started":"2021-05-22T22:35:50.333938Z","shell.execute_reply":"2021-05-22T22:35:50.339648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<img src=\"https://camo.githubusercontent.com/dd842f7b0be57140e68b2ab9cb007992acd131c48284eaf6b1aca758bfea358b/68747470733a2f2f692e696d6775722e636f6d2f52557469567a482e706e67\">\n\nI will be integrating W&B for ```visualizations``` and ```logging artifacts```!\n\n[SIIM Project on W&B Dashboard](https://wandb.ai/ruchi798/siim?workspace=user-ruchi798)","metadata":{}},{"cell_type":"code","source":"from kaggle_secrets import UserSecretsClient\nuser_secrets = UserSecretsClient()\nsecret_value_0 = user_secrets.get_secret(\"mySecret\")\n\nos.environ[\"WANDB_SILENT\"] = \"true\"\n\nCONFIG = {'competition': 'siim-fisabio-rsna', '_wandb_kernel': 'ruch'}","metadata":{"execution":{"iopub.status.busy":"2021-05-22T22:40:09.99463Z","iopub.execute_input":"2021-05-22T22:40:09.994955Z","iopub.status.idle":"2021-05-22T22:40:10.466528Z","shell.execute_reply.started":"2021-05-22T22:40:09.994926Z","shell.execute_reply":"2021-05-22T22:40:10.465549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! wandb login $secret_value_0","metadata":{"execution":{"iopub.status.busy":"2021-05-22T22:40:11.178706Z","iopub.execute_input":"2021-05-22T22:40:11.179039Z","iopub.status.idle":"2021-05-22T22:40:13.297726Z","shell.execute_reply.started":"2021-05-22T22:40:11.179008Z","shell.execute_reply":"2021-05-22T22:40:13.296867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image_level = pd.read_csv(\"../input/siim-covid19-detection/train_image_level.csv\")\ntrain_study_level = pd.read_csv(\"../input/siim-covid19-detection/train_study_level.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-05-22T22:40:16.739985Z","iopub.execute_input":"2021-05-22T22:40:16.740291Z","iopub.status.idle":"2021-05-22T22:40:16.779408Z","shell.execute_reply.started":"2021-05-22T22:40:16.740264Z","shell.execute_reply":"2021-05-22T22:40:16.778689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"| id    | unique study identifier                                      |\n|-------|--------------------------------------------------------------|\n| boxes | bounding boxes in easily-readable dictionary format          |\n| label | the correct prediction label for the provided bounding boxes |","metadata":{}},{"cell_type":"code","source":"train_image_level.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-22T22:40:20.225792Z","iopub.execute_input":"2021-05-22T22:40:20.226248Z","iopub.status.idle":"2021-05-22T22:40:20.251976Z","shell.execute_reply.started":"2021-05-22T22:40:20.226219Z","shell.execute_reply":"2021-05-22T22:40:20.251189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"| id                       | unique study identifier                                  |\n|--------------------------|----------------------------------------------------------|\n| Negative for Pneumonia   | 1 : if the study is negative for pneumonia, 0: otherwise |\n| Typical Appearance       | 1: if the study has this appearance, 0: otherwise        |\n| Indeterminate Appearance | 1: if the study has this appearance, 0: otherwise        |\n| Atypical Appearance      | 1: if the study has this appearance, 0: otherwise        |","metadata":{}},{"cell_type":"code","source":"train_study_level.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-22T22:40:21.592047Z","iopub.execute_input":"2021-05-22T22:40:21.592378Z","iopub.status.idle":"2021-05-22T22:40:21.603523Z","shell.execute_reply.started":"2021-05-22T22:40:21.592345Z","shell.execute_reply":"2021-05-22T22:40:21.602718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_directory = \"../input/siim-covid19-detection/train/\"\ntest_directory = \"../input/siim-covid19-detection/test/\"\n\ntrain_study_level['StudyInstanceUID'] = train_study_level['id'].apply(lambda x: x.replace('_study', ''))\ndel train_study_level['id']\ntrain_df = train_image_level.merge(train_study_level, on='StudyInstanceUID')","metadata":{"execution":{"iopub.status.busy":"2021-05-22T22:40:22.454876Z","iopub.execute_input":"2021-05-22T22:40:22.455179Z","iopub.status.idle":"2021-05-22T22:40:22.476743Z","shell.execute_reply.started":"2021-05-22T22:40:22.455154Z","shell.execute_reply":"2021-05-22T22:40:22.475656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-22T22:40:22.88814Z","iopub.execute_input":"2021-05-22T22:40:22.888492Z","iopub.status.idle":"2021-05-22T22:40:22.900973Z","shell.execute_reply.started":"2021-05-22T22:40:22.888462Z","shell.execute_reply":"2021-05-22T22:40:22.900019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_paths = []\n\nfor sid in tqdm(train_df['StudyInstanceUID']):\n    training_paths.append(glob.glob(os.path.join(train_directory, sid +\"/*/*\"))[0])\n\ntrain_df['path'] = training_paths","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-05-22T22:40:24.737407Z","iopub.execute_input":"2021-05-22T22:40:24.737722Z","iopub.status.idle":"2021-05-22T22:40:50.248606Z","shell.execute_reply.started":"2021-05-22T22:40:24.737694Z","shell.execute_reply":"2021-05-22T22:40:50.247755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-22T22:41:43.462138Z","iopub.execute_input":"2021-05-22T22:41:43.462456Z","iopub.status.idle":"2021-05-22T22:41:43.475226Z","shell.execute_reply.started":"2021-05-22T22:41:43.462428Z","shell.execute_reply":"2021-05-22T22:41:43.474357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Distribution of class labels","metadata":{}},{"cell_type":"code","source":"params = {'legend.fontsize': 'x-large',\n          'figure.figsize': (20, 32),\n         'axes.labelsize': 'x-large',\n         'axes.titlesize':'x-large',\n         'xtick.labelsize':'x-large',\n         'ytick.labelsize':'x-large'}\npylab.rcParams.update(params)\n\nfig, ax = plt.subplots(4,2)\nsns.kdeplot(train_df[\"Negative for Pneumonia\"], shade=True,ax=ax[0,0],color=\"#ffb4a2\")\nax[0,0].set_title(\"Negative for Pneumonia Distribution\",font=\"Serif\", fontsize=20,weight=\"bold\")\nsns.countplot(x = train_df[\"Negative for Pneumonia\"], ax=ax[0,1],color=\"#ffb4a2\")\nax[0,1].set_title(\"Negative for Pneumonia Distribution\",font=\"Serif\", fontsize=20,weight=\"bold\")\n\nsns.kdeplot(train_df[\"Typical Appearance\"], shade=True,ax=ax[1,0],color=\"#e5989b\")\nax[1,0].set_title(\"Typical Appearance Distribution\",font=\"Serif\", fontsize=20,weight=\"bold\")\nsns.countplot(x = train_df[\"Typical Appearance\"], ax=ax[1,1],color=\"#e5989b\")\nax[1,1].set_title(\"Typical Appearance Distribution\",font=\"Serif\", fontsize=20,weight=\"bold\")\n\nsns.kdeplot(train_df[\"Indeterminate Appearance\"], shade=True,ax=ax[2,0],color=\"#b5838d\")\nax[2,0].set_title(\"Indeterminate Appearance Distribution\",font=\"Serif\", fontsize=20,weight=\"bold\")\nsns.countplot(x = train_df[\"Indeterminate Appearance\"], ax=ax[2,1],color=\"#b5838d\")\nax[2,1].set_title(\"Indeterminate Appearance Distribution\",font=\"Serif\", fontsize=20,weight=\"bold\")\n\nsns.kdeplot(train_df[\"Atypical Appearance\"], shade=True,ax=ax[3,0],color=\"#6d6875\")\nax[3,0].set_title(\"Atypical Appearance Distribution\",font=\"Serif\", fontsize=20,weight=\"bold\")\nsns.countplot(x = train_df[\"Atypical Appearance\"], ax=ax[3,1],color=\"#6d6875\")\nax[3,1].set_title(\"Atypical Appearance Distribution\",font=\"Serif\", fontsize=20,weight=\"bold\")\n\nfig.subplots_adjust(wspace=0.2, hspace=0.4, top=0.93)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-22T22:41:44.468367Z","iopub.execute_input":"2021-05-22T22:41:44.468661Z","iopub.status.idle":"2021-05-22T22:41:45.724995Z","shell.execute_reply.started":"2021-05-22T22:41:44.468637Z","shell.execute_reply":"2021-05-22T22:41:45.724109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#====== Function to plot wandb bar chart ======\ndef plot_wb_bar(df,col1,col2): \n    run = wandb.init(project='siim', job_type='image-visualization',name=col1,config = CONFIG)\n    \n    dt = [[label, val] for (label, val) in zip(df[col1], df[col2])]\n    table = wandb.Table(data=dt, columns = [col1,col2])\n    wandb.log({col1 : wandb.plot.bar(table, col1,col2,title=col1)})\n\n    run.finish()\n    \n#====== Function to create a dataframe of value counts ======\ndef count_values(col):\n    df = pd.DataFrame(train_df[col].value_counts().reset_index().values,columns=[col, \"counts\"])\n    return df\n\nplot_wb_bar(count_values(\"Negative for Pneumonia\"),\"Negative for Pneumonia\", 'counts')\nplot_wb_bar(count_values(\"Typical Appearance\"),\"Typical Appearance\", 'counts')\nplot_wb_bar(count_values(\"Indeterminate Appearance\"),\"Indeterminate Appearance\", 'counts')\nplot_wb_bar(count_values(\"Atypical Appearance\"),\"Atypical Appearance\", 'counts')","metadata":{"execution":{"iopub.status.busy":"2021-05-22T22:43:02.77929Z","iopub.execute_input":"2021-05-22T22:43:02.77965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# DICOM data","metadata":{}},{"cell_type":"markdown","source":"### Image and Metadata","metadata":{}},{"cell_type":"code","source":"voi_lut=True\nfix_monochrome=True\n\ndef dicom_dataset_to_dict(filename,func):\n    \"\"\"Credit: https://github.com/pydicom/pydicom/issues/319\n               https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    \"\"\"\n    \n    dicom_header = dicom.dcmread(filename) \n    \n    #====== DICOM FILE DATA ======\n    dicom_dict = {}\n    repr(dicom_header)\n    for dicom_value in dicom_header.values():\n        if dicom_value.tag == (0x7fe0, 0x0010):\n            #discard pixel data\n            continue\n        if type(dicom_value.value) == dicom.dataset.Dataset:\n            dicom_dict[dicom_value.name] = dicom_dataset_to_dict(dicom_value.value)\n        else:\n            v = _convert_value(dicom_value.value)\n            dicom_dict[dicom_value.name] = v\n      \n    del dicom_dict['Pixel Representation']\n    \n    if func!='metadata_df':\n        #====== DICOM IMAGE DATA ======\n        # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \"human-friendly\" view\n        if voi_lut:\n            data = apply_voi_lut(dicom_header.pixel_array, dicom_header)\n        else:\n            data = dicom_header.pixel_array\n        # depending on this value, X-ray may look inverted - fix that:\n        if fix_monochrome and dicom_header.PhotometricInterpretation == \"MONOCHROME1\":\n            data = np.amax(data) - data\n        data = data - np.min(data)\n        data = data / np.max(data)\n        modified_image_data = (data * 255).astype(np.uint8)\n    \n        return dicom_dict, modified_image_data\n    \n    else:\n        return dicom_dict\n\ndef _sanitise_unicode(s):\n    return s.replace(u\"\\u0000\", \"\").strip()\n\ndef _convert_value(v):\n    t = type(v)\n    if t in (list, int, float):\n        cv = v\n    elif t == str:\n        cv = _sanitise_unicode(v)\n    elif t == bytes:\n        s = v.decode('ascii', 'replace')\n        cv = _sanitise_unicode(s)\n    elif t == dicom.valuerep.DSfloat:\n        cv = float(v)\n    elif t == dicom.valuerep.IS:\n        cv = int(v)\n    else:\n        cv = repr(v)\n    return cv\n\nfor filename in train_df.path[0:5]:\n    df, img_array = dicom_dataset_to_dict(filename, 'fetch_both_values')\n    \n    fig, ax = plt.subplots(1, 2, figsize=[15, 8])\n    ax[0].imshow(img_array, cmap=plt.cm.gray)\n    ax[1].imshow(img_array, cmap=plt.cm.plasma)    \n    plt.show()\n    \n    pprint.pprint(df)","metadata":{"execution":{"iopub.status.busy":"2021-05-22T05:31:04.292683Z","iopub.execute_input":"2021-05-22T05:31:04.293021Z","iopub.status.idle":"2021-05-22T05:31:21.263507Z","shell.execute_reply.started":"2021-05-22T05:31:04.292985Z","shell.execute_reply":"2021-05-22T05:31:21.262653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dicom_data_list = []\n# for filename in train_df.path:\n#     try:\n#         data_di = dicom_dataset_to_dict(filename,'metadata_df')\n#         dicom_data_list.append(data_di)\n    \n#     except:\n#         continue\n\n# dicom_data_df = pd.DataFrame(dicom_data_list) \n# dicom_data_df\n\n# #====== Saving to csv files and creating artifacts ======\n# dicom_data_df.to_csv(\"dicom_metadata.csv\")\n\n# run = wandb.init(project='siim', name='dicom_metadata')\n\n# artifact = wandb.Artifact('dicom_metadata', type='dataset')\n\n# #====== Add a file to the artifact's contents ======\n# artifact.add_file(\"dicom_metadata.csv\")\n\n# #====== Save the artifact version to W&B and mark it as the output of this run ====== \n# run.log_artifact(artifact)\n\n# run.finish()","metadata":{"execution":{"iopub.status.busy":"2021-05-22T05:31:21.264628Z","iopub.execute_input":"2021-05-22T05:31:21.264943Z","iopub.status.idle":"2021-05-22T05:31:21.270531Z","shell.execute_reply.started":"2021-05-22T05:31:21.264913Z","shell.execute_reply":"2021-05-22T05:31:21.269455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"run = wandb.init(project='siim', config = CONFIG)\nartifact = run.use_artifact('ruchi798/siim/dicom_metadata:v1', type='dataset')\nartifact_dir = artifact.download()\nrun.finish()\n\npath = os.path.join(artifact_dir,\"dicom_metadata.csv\")\nmetadata = pd.read_csv(path)\nmetadata = metadata.drop(columns=[\"Unnamed: 0\"])\nmetadata.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-22T05:31:21.272493Z","iopub.execute_input":"2021-05-22T05:31:21.272886Z","iopub.status.idle":"2021-05-22T05:31:32.384269Z","shell.execute_reply.started":"2021-05-22T05:31:21.272854Z","shell.execute_reply":"2021-05-22T05:31:32.383502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"A snapshot of the newly created artifact:\n<img src=\"https://i.imgur.com/QVHNpcB.png\">","metadata":{}},{"cell_type":"code","source":"def label_sizes(col):\n    labels = metadata[col].value_counts().index\n    sizes = metadata[col].value_counts()\n    uc = metadata[col].nunique()\n    return labels, sizes, uc\n\ndef plot_pie(col1,col2,c1,c2):\n    fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(24,10))\n    axs = [ax1, ax2]\n    \n    labels, sizes, uc = label_sizes(col1)\n    explode = (0.05,)*uc\n    \n    if col1 == \"De-identification Method\":\n        labels = list(map(lambda b: b.replace(\"CTP Default:  based on DICOM PS3.15 AnnexE. Details in 0012,0064\",\"CTP Default:  based on DICOM PS3.15\"), labels))\n    \n    \n    ax1.pie(sizes, explode=explode, colors=c1, startangle=60, labels=labels,autopct='%1.0f%%', pctdistance=0.6)\n    ax1.add_artist(plt.Circle((0,0),0.4,fc='white'))\n    ax1.set_title(col1 + \" Distribution\",weight=\"bold\")\n\n    labels, sizes, uc = label_sizes(col2)\n    explode = (0.05,)*uc\n    ax2.pie(sizes, explode=explode, colors=c2, startangle=60, labels=labels,autopct='%1.0f%%', pctdistance=0.6)\n    ax2.add_artist(plt.Circle((0,0),0.4,fc='white'))\n    ax2.set_title(col2 + \" Distribution\",weight=\"bold\")\n    \n    plt.show()\n    \nplot_pie(\"Modality\",\"Photometric Interpretation\",['#5C8DFF','#abc4ff'],['#05979E','#87F5FB'])\n\nplt.figure(figsize=(16, 8))\nsns.countplot(y=\"Body Part Examined\",data=metadata,linewidth=3,palette=\"PRGn\")\nplt.title(\"Body Part Examined Distribution\",font=\"Serif\", size = 20,weight=\"bold\")\nplt.show()\n\nplt.figure(figsize=(16, 8))\nsns.countplot(y=\"Private Creator\",data=metadata,linewidth=3,palette=['#F9ADA0','#F9627D',\"#6DAEDB\"])\nplt.title(\"Private Creator Distribution\",font=\"Serif\", size = 20, weight=\"bold\")\nplt.show()\n\nplot_pie(\"De-identification Method\",\"Patient's Sex\",['#F3C98B',\"#fff3b0\",'#DE8E17'],['#E6C4E9','#C77ACD'])","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-05-22T05:31:32.385551Z","iopub.execute_input":"2021-05-22T05:31:32.386071Z","iopub.status.idle":"2021-05-22T05:31:33.367282Z","shell.execute_reply.started":"2021-05-22T05:31:32.386033Z","shell.execute_reply":"2021-05-22T05:31:33.366103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# initializing the run\nrun = wandb.init(project=\"siim\",\n                 job_type=\"upload\",\n                 config = CONFIG\n                 )\n\n# creating an artifact\nartifact = wandb.Artifact(name=\"dicom_metadata_image\", type=\"raw_data\")\n\n# setting up a WandB Table object to hold the dataset\ncolumns = ['image',\"Body Part Examined\",\"Image Type\",\"Modality\",\"Patient's Name\",\"Patient ID\",\"Patient's Sex\",\"Study Instance UID\"]\ntable = wandb.Table(\n    columns=columns\n)\n\nfor filename in train_df.path[0:5]:\n    data_di, img_array = dicom_dataset_to_dict(filename,'fetch_both_values')\n    \n    body_part_examined = data_di.get(\"Body Part Examined\")\n    img_type = data_di.get('Image Type')\n    modality = data_di.get(\"Modality\")\n    p_name = data_di.get(\"Patient's Name\")\n    p_id = data_di.get(\"Patient ID\")\n    p_gender = data_di.get(\"Patient's Sex\")\n    study_inst_uid = data_di.get(\"Study Instance UID\")\n    \n    img_object = Image.fromarray(img_array)\n    # raw image\n    raw_img = wandb.Image(img_object)\n\n    # adding a row to the table\n    row = [raw_img,body_part_examined,img_type,modality,p_name,p_id,p_gender,study_inst_uid]\n    table.add_data(*row)\n       \n# adding the table to the artifact\nartifact.add(table, \"dicom_examples\")\n   \n# logging the artifact\nrun.log_artifact(artifact)\n\nrun.finish()","metadata":{"execution":{"iopub.status.busy":"2021-05-22T05:31:33.369202Z","iopub.execute_input":"2021-05-22T05:31:33.369618Z","iopub.status.idle":"2021-05-22T05:31:54.869471Z","shell.execute_reply.started":"2021-05-22T05:31:33.369575Z","shell.execute_reply":"2021-05-22T05:31:54.868364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Visualizing the DICOM data in a W&B table: ","metadata":{}},{"cell_type":"markdown","source":"<img src = \"https://i.imgur.com/SeAidj8.gif\">","metadata":{}},{"cell_type":"markdown","source":"### Images and Classes","metadata":{}},{"cell_type":"code","source":"# initializing the run\nrun = wandb.init(project=\"siim\",\n                 job_type=\"upload\",\n                 config = CONFIG\n                 )\n\n# creating an artifact \nartifact = wandb.Artifact(name=\"dicom_images\", type=\"raw_data\")\n\n# setting up a WandB Table object to hold the dataset\ncolumns=[\"dicom image\", \"class\"]\n\ntable = wandb.Table(\n    columns=columns\n)\n\nclasses = ['Negative for Pneumonia','Typical Appearance', 'Indeterminate Appearance', 'Atypical Appearance']\nfor siim_class in classes:\n    print(siim_class)\n    for _, row in train_df[train_df[siim_class]==1].iloc[:2].iterrows():\n        filename = row['path']\n        df, img_array = dicom_dataset_to_dict(filename,'fetch_both_values')\n        \n        fig, ax = plt.subplots(1, 2, figsize=[15, 8])\n        ax[0].imshow(img_array, cmap=plt.cm.gray)\n        ax[1].imshow(img_array, cmap=plt.cm.plasma)   \n        plt.show()\n        \n        img_object = Image.fromarray(img_array)\n        \n        # raw image\n        raw_img = wandb.Image(img_object)\n\n        # adding a row to the table\n        row = [raw_img,siim_class]\n        table.add_data(*row)\n        \n# adding the table to the artifact\nartifact.add(table, \"raw_examples\")\n    \n# logging the artifact\nrun.log_artifact(artifact)\n\nrun.finish()","metadata":{"execution":{"iopub.status.busy":"2021-05-22T05:31:54.871133Z","iopub.execute_input":"2021-05-22T05:31:54.871455Z","iopub.status.idle":"2021-05-22T05:32:42.279268Z","shell.execute_reply.started":"2021-05-22T05:31:54.871425Z","shell.execute_reply":"2021-05-22T05:32:42.277984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can interact with the W&B table by specifying filters on any column to **limit the visible rows down to only rows that match**!\n\nHere I've filtered the table to see only those images that have the class label as ```Atypical Appearance``` or ```Indeterminate Appearance```.","metadata":{}},{"cell_type":"markdown","source":"<img src = \"https://i.imgur.com/Cbcn9nP.gif\">","metadata":{}},{"cell_type":"markdown","source":"Super thankful to @[xhlulu](https://www.kaggle.com/xhlulu) for converting [dicom image data to jpg files](https://www.kaggle.com/xhlulu/siim-covid-19-convert-to-jpg-256px)! ⚡","metadata":{}},{"cell_type":"code","source":"train_jpg_directory = '../input/siim-covid19-resized-to-256px-jpg/train'\ntest_jpg_directory = '../input/siim-covid19-resized-to-256px-jpg/test'\n\ndef getImagePaths(path):\n    image_names = []\n    for dirname, _, filenames in os.walk(path):\n        for filename in filenames:\n            fullpath = os.path.join(dirname, filename)\n            image_names.append(fullpath)\n    return image_names\n\ntrain_images_path = getImagePaths(train_jpg_directory)\ntest_images_path = getImagePaths(test_jpg_directory)\n\nprint(f\"{y_}Number of train images: {g_} {len(train_images_path)}\\n\")\nprint(f\"{y_}Number of test images: {g_} {len(test_images_path)}\\n\")\n\ndef getShape(data, images_paths):\n    shape = cv2.imread(images_paths[0]).shape\n    for image_path in images_paths:\n        image_shape=cv2.imread(image_path).shape\n        if (image_shape!=shape):\n            return data +\" - Different image shape\"\n        else:\n            return data +\" - Same image shape \" + str(shape)","metadata":{"execution":{"iopub.status.busy":"2021-05-22T05:32:42.281252Z","iopub.execute_input":"2021-05-22T05:32:42.281692Z","iopub.status.idle":"2021-05-22T05:32:46.304552Z","shell.execute_reply.started":"2021-05-22T05:32:42.281644Z","shell.execute_reply":"2021-05-22T05:32:46.303647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"run = wandb.init(project='siim', name='count',config = CONFIG)\n\nwandb.log({'Training samples': len(train_images_path) , \n           'Test samples': len(test_images_path) \n          })\n\nrun.finish()","metadata":{"execution":{"iopub.status.busy":"2021-05-22T05:32:46.305769Z","iopub.execute_input":"2021-05-22T05:32:46.306285Z","iopub.status.idle":"2021-05-22T05:32:54.585093Z","shell.execute_reply.started":"2021-05-22T05:32:46.306248Z","shell.execute_reply":"2021-05-22T05:32:54.584166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Checking if images in each directory have the same shape","metadata":{}},{"cell_type":"code","source":"getShape('train',train_images_path)","metadata":{"execution":{"iopub.status.busy":"2021-05-22T05:32:54.586687Z","iopub.execute_input":"2021-05-22T05:32:54.587051Z","iopub.status.idle":"2021-05-22T05:32:54.62638Z","shell.execute_reply.started":"2021-05-22T05:32:54.587019Z","shell.execute_reply":"2021-05-22T05:32:54.62497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"getShape('test',test_images_path)","metadata":{"execution":{"iopub.status.busy":"2021-05-22T05:32:54.627693Z","iopub.execute_input":"2021-05-22T05:32:54.62804Z","iopub.status.idle":"2021-05-22T05:32:54.645104Z","shell.execute_reply.started":"2021-05-22T05:32:54.628009Z","shell.execute_reply":"2021-05-22T05:32:54.644128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Augmentation","metadata":{}},{"cell_type":"code","source":"def plot_augmentations(images, titles, sup_title):\n    fig, axes = plt.subplots(figsize=(20, 16), nrows=3, ncols=4, squeeze=False)\n    \n    for indx, (img, title) in enumerate(zip(images, titles)):\n        axes[indx // 4][indx % 4].imshow(img)\n        axes[indx // 4][indx % 4].set_title(title, fontsize=15)\n        \n    plt.tight_layout()\n    fig.suptitle(sup_title, fontsize = 20)\n    fig.subplots_adjust(wspace=0.2, hspace=0.2, top=0.93)\n    plt.show()\n    \ndef augment(paths, data):\n    \n    # list of albumentations\n    albumentations = [A.RandomSunFlare(p=1), A.RandomFog(p=1), A.RandomBrightness(p=1),\n                              A.RandomCrop(p=1,height = 128, width = 128), A.Rotate(p=1, limit=90),\n                              A.RGBShift(p=1), A.RandomSnow(p=1),\n                              A.HorizontalFlip(p=1), A.VerticalFlip(p=1), A.RandomContrast(limit = 0.5,p = 1),\n                              A.HueSaturationValue(p=1,hue_shift_limit=20, sat_shift_limit=30, val_shift_limit=50)]\n    \n    # image titles\n    titles = [\"RandomSunFlare\",\"RandomFog\",\"RandomBrightness\",\n                       \"RandomCrop\",\"Rotate\", \"RGBShift\", \"RandomSnow\",\"HorizontalFlip\", \"VerticalFlip\", \"RandomContrast\",\"HSV\"]\n    \n    for i in paths:\n        image_path = i\n        \n        # getting image name from path\n        image_name = image_path.split(\"/\")[4].split(\".\")[0]\n        \n        # reading image\n        image = cv2.imread(image_path)\n\n        # list of images\n        images = []\n        \n        # creating image augmentations\n        for augmentation_type in albumentations:\n            augmented_img = augmentation_type(image = image)['image']\n            images.append(augmented_img)\n\n        # original image\n        titles.insert(0, \"Original\")\n        images.insert(0,image)  \n        \n        sup_title = \"Image Augmentation for \" + data + \" - \" + image_name\n        plot_augmentations(images, titles, sup_title)\n        \n        titles.remove(\"Original\")","metadata":{"execution":{"iopub.status.busy":"2021-05-22T05:32:54.646451Z","iopub.execute_input":"2021-05-22T05:32:54.647011Z","iopub.status.idle":"2021-05-22T05:32:54.662021Z","shell.execute_reply.started":"2021-05-22T05:32:54.646969Z","shell.execute_reply":"2021-05-22T05:32:54.660938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data Augmentation (train samples)","metadata":{}},{"cell_type":"code","source":"augment(train_images_path[0:2],'train')","metadata":{"execution":{"iopub.status.busy":"2021-05-22T05:32:54.663578Z","iopub.execute_input":"2021-05-22T05:32:54.664178Z","iopub.status.idle":"2021-05-22T05:32:59.333977Z","shell.execute_reply.started":"2021-05-22T05:32:54.664125Z","shell.execute_reply":"2021-05-22T05:32:59.333127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data Augmentation (test samples)","metadata":{}},{"cell_type":"code","source":"augment(train_images_path[0:2],'test')","metadata":{"execution":{"iopub.status.busy":"2021-05-22T05:32:59.335218Z","iopub.execute_input":"2021-05-22T05:32:59.335509Z","iopub.status.idle":"2021-05-22T05:33:04.083879Z","shell.execute_reply.started":"2021-05-22T05:32:59.335479Z","shell.execute_reply":"2021-05-22T05:33:04.082916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I've created a [dataset](https://www.kaggle.com/ruchi798/siimfisabiorsna-covid19-detection-augmented) of image augmentations for all the training and testing images as well 🥳","metadata":{}},{"cell_type":"markdown","source":"This is what my [project](https://wandb.ai/ruchi798/siim?workspace=user-ruchi798) looks like on the W&B dashboard ⬇️\n<img src=\"https://i.imgur.com/lFIrsJT.png\">","metadata":{}},{"cell_type":"markdown","source":"Work in Progress ⏳","metadata":{}}]}