{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"9ce9d796-87a6-4c6a-bc2e-514046db491b","_cell_guid":"f9298925-72a4-455e-bf71-c745bc618337","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:18:50.891102Z","iopub.execute_input":"2021-06-10T12:18:50.891411Z","iopub.status.idle":"2021-06-10T12:18:50.899711Z","shell.execute_reply.started":"2021-06-10T12:18:50.891335Z","shell.execute_reply":"2021-06-10T12:18:50.898868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"280ef670-e9bd-4f53-a52f-f00c990b94ac","_cell_guid":"9c8737f0-b9b6-47f1-abec-484f47008fc2","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:18:50.900812Z","iopub.execute_input":"2021-06-10T12:18:50.901131Z","iopub.status.idle":"2021-06-10T12:18:50.920999Z","shell.execute_reply.started":"2021-06-10T12:18:50.901107Z","shell.execute_reply":"2021-06-10T12:18:50.920575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pydicom as dicom\nimport matplotlib\nimport matplotlib.pylab as plt\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport glob\nimport cv2","metadata":{"_uuid":"d10397bc-aa6f-4645-956f-6832c3065255","_cell_guid":"1f9940ff-0f40-49aa-b6a3-51901ccdc0a2","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:36:10.132426Z","iopub.execute_input":"2021-06-10T12:36:10.132833Z","iopub.status.idle":"2021-06-10T12:36:10.145926Z","shell.execute_reply.started":"2021-06-10T12:36:10.132799Z","shell.execute_reply":"2021-06-10T12:36:10.144942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class color:\n    PURPLE = '\\033[1;35;48m'\n    CYAN = '\\033[1;36;48m'\n    BOLD = '\\033[1;37;48m'\n    BLUE = '\\033[1;34;48m'\n    GREEN = '\\033[1;32;48m'\n    YELLOW = '\\033[1;33;48m'\n    RED = '\\033[1;31;48m'\n    BLACK = '\\033[1;30;48m'\n    UNDERLINE = '\\033[4;37;48m'\n    MAGENTA = \"\\033[35m\"\n    WHITE = \"\\033[97m\"\n    BOLD = '\\033[1m' + '\\033[93m'\n    END = '\\033[0m'\n    \n    \nplt.rcParams['figure.figsize'] = [9, 7]\npd.options.display.max_columns = None","metadata":{"_uuid":"9cb6ef13-d7be-4a11-9b98-0da8195e36a1","_cell_guid":"60734ba5-e905-4c75-8469-236c4f87e7ee","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T13:11:53.039226Z","iopub.execute_input":"2021-06-10T13:11:53.039568Z","iopub.status.idle":"2021-06-10T13:11:53.045901Z","shell.execute_reply.started":"2021-06-10T13:11:53.039524Z","shell.execute_reply":"2021-06-10T13:11:53.044994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !conda install -c conda-forge gdcm -y      # just commented out to save commit time","metadata":{"_uuid":"a208f44b-6f74-49ca-ba32-4e7b285445e4","_cell_guid":"ff7541c6-e74c-4b41-bcf2-6cfb06f52f7e","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:18:51.113713Z","iopub.execute_input":"2021-06-10T12:18:51.114105Z","iopub.status.idle":"2021-06-10T12:18:51.130497Z","shell.execute_reply.started":"2021-06-10T12:18:51.114073Z","shell.execute_reply":"2021-06-10T12:18:51.129994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_process_csv():\n    ''' Read the files from disk, remove redundant strings from columns, merge both\n        the dfs to form 1 and few pre-processing steps.\n    '''\n    \n    train_image = pd.read_csv('../input/siim-covid19-detection/train_image_level.csv')\n    train_image['id'] = train_image['id'].str.split('_', expand = True)[0]\n    \n    train_study = pd.read_csv('../input/siim-covid19-detection/train_study_level.csv')\n    train_study['id'] = train_study['id'].str.split('_', expand = True)[0]\n    \n    train = pd.merge(train_image, train_study, left_on='StudyInstanceUID', right_on='id')\n    \n    train.pop('StudyInstanceUID')\n    \n    train.rename(columns = {\"id_x\" : \"image_id\", \"id_y\" : \"study_id\"}, inplace = True)\n    \n    clmns = train.columns.tolist()\n    clmns = [clmns[0], clmns[3], clmns[1], clmns[2]] + clmns[4:]\n    train = train[clmns]\n    \n    return train_image, train_study, train","metadata":{"_uuid":"affa86dc-6f6a-469a-a72b-7d04081af3fa","_cell_guid":"23a827f6-e7d7-4db9-8c5a-42cf0235cf58","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:18:51.245454Z","iopub.execute_input":"2021-06-10T12:18:51.245906Z","iopub.status.idle":"2021-06-10T12:18:51.252437Z","shell.execute_reply.started":"2021-06-10T12:18:51.245872Z","shell.execute_reply":"2021-06-10T12:18:51.251925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image, train_study, train = get_process_csv()\ntrain_len = len(train)\ntrain.info()","metadata":{"_uuid":"d4934791-84f8-402a-aa66-598064f806f6","_cell_guid":"1f62c681-0eef-4584-b039-53fe5a37a2fb","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:18:51.275144Z","iopub.execute_input":"2021-06-10T12:18:51.275408Z","iopub.status.idle":"2021-06-10T12:18:51.340398Z","shell.execute_reply.started":"2021-06-10T12:18:51.275386Z","shell.execute_reply":"2021-06-10T12:18:51.339354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe(include = 'all').T","metadata":{"_uuid":"66548169-edf8-407a-b85e-fa8bbc823441","_cell_guid":"9ed141f9-857e-4f78-ac24-72db98713eb9","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:18:51.379931Z","iopub.execute_input":"2021-06-10T12:18:51.380289Z","iopub.status.idle":"2021-06-10T12:18:51.424018Z","shell.execute_reply.started":"2021-06-10T12:18:51.380266Z","shell.execute_reply":"2021-06-10T12:18:51.422931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_images_path = '../input/siim-covid19-detection/train'\nimage_files_path = glob.glob(f'{train_images_path}/**/*.dcm', recursive=True)  # extract images from the directory","metadata":{"_uuid":"a7b4cae9-5435-47b4-9378-f36a51d4d417","_cell_guid":"27e45ef4-4b09-4178-ab28-9e3ce96b13cb","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:18:51.425266Z","iopub.execute_input":"2021-06-10T12:18:51.425476Z","iopub.status.idle":"2021-06-10T12:19:00.904546Z","shell.execute_reply.started":"2021-06-10T12:18:51.425454Z","shell.execute_reply":"2021-06-10T12:19:00.903591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_image_path_col(train):\n    '''Add directory path of the image to the dataframe\n    '''\n    \n    train.insert(2, 'image_path', np.nan)\n    for image_path in image_files_path:\n\n        image_id = image_path.split('/')[-1].split('.')[0]\n        index = train[train['image_id'] == image_id].index\n        train.at[index, 'image_path'] = image_path\n        \n    return train","metadata":{"_uuid":"26c24592-152b-40cf-8464-07bca98d4be7","_cell_guid":"71dc3372-fb5c-4a12-ad56-3bb7c7285ccd","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:19:00.905635Z","iopub.execute_input":"2021-06-10T12:19:00.905834Z","iopub.status.idle":"2021-06-10T12:19:00.910088Z","shell.execute_reply.started":"2021-06-10T12:19:00.905813Z","shell.execute_reply":"2021-06-10T12:19:00.909289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = add_image_path_col(train)\ntrain.head()","metadata":{"_uuid":"bf2686db-df6d-489d-aa91-856772b50b4d","_cell_guid":"8258f4f9-c35f-44cc-8c74-e0352ad212f9","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:19:00.911141Z","iopub.execute_input":"2021-06-10T12:19:00.911344Z","iopub.status.idle":"2021-06-10T12:19:08.886260Z","shell.execute_reply.started":"2021-06-10T12:19:00.911322Z","shell.execute_reply":"2021-06-10T12:19:08.885514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def dicom2array(path, voi_lut=True, fix_monochrome=True):\n    # convert dcm type images into numpy arrays\n    \n    dcm = dicom.read_file(path)\n    if voi_lut:\n        try:\n            data = apply_voi_lut(dcm.pixel_array, dcm)\n        except RuntimeError:\n            print('An error occured while de-compressing the file from dicom to array')\n            return [1]\n    else:\n        data = dcm.pixel_array\n    if fix_monochrome and dcm.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data","metadata":{"_uuid":"a89f306b-5e23-460d-86c1-b8c014343419","_cell_guid":"244bb972-abce-4cb3-8933-8ac56db6b2f3","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:19:08.887148Z","iopub.execute_input":"2021-06-10T12:19:08.887358Z","iopub.status.idle":"2021-06-10T12:19:08.892520Z","shell.execute_reply.started":"2021-06-10T12:19:08.887335Z","shell.execute_reply":"2021-06-10T12:19:08.891813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_data_cols = ['SpecificCharacterSet', 'SOPClassUID', 'SOPInstanceUID', 'StudyDate', 'StudyTime', 'AccessionNumber',\n            'Modality', 'PatientName', 'PatientID', 'PatientSex', 'BodyPartExamined', 'PhotometricInterpretation']\n\ndef add_meta_cols(train, meta_cols):\n    '''Add meta data from dcm files to the dataframe\n    '''\n    train[meta_cols] = np.nan\n    \n    for image_path in image_files_path:\n        image_data = dicom.dcmread(image_path)\n        \n        index = train[train['image_path'] == image_path].index\n\n        for col in meta_cols:\n            train.at[index, col] = str(image_data.get(col))\n            \n    return train","metadata":{"_uuid":"a5bd7731-17d3-4f8a-882b-c6c13ebdcd42","_cell_guid":"e7ea9df6-97ec-4fdb-9af3-1d5b76e3f0fa","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:19:08.893315Z","iopub.execute_input":"2021-06-10T12:19:08.893515Z","iopub.status.idle":"2021-06-10T12:19:08.906589Z","shell.execute_reply.started":"2021-06-10T12:19:08.893495Z","shell.execute_reply":"2021-06-10T12:19:08.906005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train = add_meta_cols(train, meta_data_cols)  # will take around 20 min to run this code :(\n# train.to_csv('./train_v1.csv', index=False)\ntrain = pd.read_csv('../input/train-v1/train_v1.csv')\ntrain = train.drop('boxes', 1)  # remove the boxes column from dataset\ntrain.head()","metadata":{"_uuid":"a98b6ec3-25b2-483f-8ebf-46a3b7441254","_cell_guid":"2f477bb9-7abe-43fc-ab49-3648cff53cae","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:19:08.908592Z","iopub.execute_input":"2021-06-10T12:19:08.908979Z","iopub.status.idle":"2021-06-10T12:19:09.017165Z","shell.execute_reply.started":"2021-06-10T12:19:08.908952Z","shell.execute_reply":"2021-06-10T12:19:09.016547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_bbox(path, label, part):\n    '''Plot the dcm image with bounding boxes\n    '''\n    \n    thickness = 5\n    img = dicom2array(path)\n    \n    if len(img) == 1:\n        print('Cannot display this image, try with different one.')\n        return \n    \n    plt.figure(figsize = (10, 8))\n    \n    if 'none' in label:\n        plt.title(part)\n        plt.imshow(img, cmap = 'bone')\n        \n    else:\n        count = label.count('opacity')\n        label = label.split()\n        label = [float(val) for val in label if val not in ('opacity', '1')]\n        \n        for k in range(count):\n            i = k * 4\n            j = (k + 1) * 4\n            box = label[i:j]\n            cv2.rectangle(img, box, color = [255, 0, 0], thickness = thickness)\n        \n        plt.title(part)\n        plt.imshow(img, cmap = 'bone')","metadata":{"_uuid":"136e1d51-a6c0-4c1f-9c2e-1b819eb777ee","_cell_guid":"eb079414-8718-47f7-bb99-862b17d975dd","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:35:48.416738Z","iopub.execute_input":"2021-06-10T12:35:48.416992Z","iopub.status.idle":"2021-06-10T12:35:48.423316Z","shell.execute_reply.started":"2021-06-10T12:35:48.416970Z","shell.execute_reply":"2021-06-10T12:35:48.422631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"index = np.random.randint(train_len)\nplot_bbox(train.loc[index, 'image_path'], train.loc[index, 'label'], train.loc[index, 'BodyPartExamined'])","metadata":{"_uuid":"20f91552-d093-4510-9190-3ed80de51da2","_cell_guid":"a00e5982-59f5-4290-a5a3-8c8439b46ba0","collapsed":true,"jupyter":{"outputs_hidden":true},"execution":{"iopub.status.busy":"2021-06-10T12:37:54.362202Z","iopub.execute_input":"2021-06-10T12:37:54.362476Z","iopub.status.idle":"2021-06-10T12:37:55.707012Z","shell.execute_reply.started":"2021-06-10T12:37:54.362453Z","shell.execute_reply":"2021-06-10T12:37:55.706156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{"_uuid":"4ef22237-bf1f-4964-b0e4-d5454131d6fc","_cell_guid":"0be835ef-c095-489c-aa81-1572bcb87e88","trusted":true}},{"cell_type":"markdown","source":"## How many images are there in the dataset?","metadata":{"_uuid":"84af65cd-a685-421f-8bd4-0d7276283553","_cell_guid":"b348505d-0b32-4400-866e-7580007085de","trusted":true}},{"cell_type":"code","source":"print(color.BOLD + 'Images in the dataset:' + color.END, train['image_id'].nunique())","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:20:51.808570Z","iopub.execute_input":"2021-06-10T13:20:51.808828Z","iopub.status.idle":"2021-06-10T13:20:51.815492Z","shell.execute_reply.started":"2021-06-10T13:20:51.808807Z","shell.execute_reply":"2021-06-10T13:20:51.814523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## How many study levels does the dataset consist?","metadata":{"_uuid":"4b4479c9-f458-42a8-8065-bc0a31add004","_cell_guid":"1fa22628-7c89-4300-87a5-fe49b429a746","trusted":true}},{"cell_type":"code","source":"print(color.BOLD + 'study Level in the dataset:' + color.END, train['study_id'].nunique())","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:22:09.679446Z","iopub.execute_input":"2021-06-10T13:22:09.679730Z","iopub.status.idle":"2021-06-10T13:22:09.685104Z","shell.execute_reply.started":"2021-06-10T13:22:09.679708Z","shell.execute_reply":"2021-06-10T13:22:09.684590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## How is study level distributed?","metadata":{"_uuid":"a381b6de-ef24-481d-a186-faee75ddce21","_cell_guid":"7e146555-21e9-4ea4-91c6-652806d0f664","trusted":true}},{"cell_type":"code","source":"a = train['study_id'].value_counts().value_counts().sort_index()\nplot = a.plot(kind = 'bar')\nplt.xticks(rotation = 0)\nplt.xlabel('Number of images in the study', fontsize = 12)\nplt.ylabel('Number of study', fontsize = 12)\n\nfor bar in plot.patches:\n    plot.annotate(bar.get_height(), \n               (bar.get_x() + bar.get_width() / 2, \n                bar.get_height()), ha='center', va='center',\n               size=14, xytext=(0, 8),\n               textcoords='offset points')\n\nfor index in a.index:\n    print(f'Studies with {index} images: {a[index]}.',\n          f'It has {round(a[index]/train_len * 100, 2)}% images of the dataset')\n\nprint(f'\\n\\nWe can see that about 92% of images in the dataset belong to unqiue studies. Most of the studies consist of just a single image.')\nprint(f'{color.BOLD}A study at maximum has 9 images.\\nA study at minimum has 1 image{color.END}.\\n')","metadata":{"_uuid":"2e23556b-dd0c-495a-8c4f-ae33b79d257f","_cell_guid":"a9345a7f-efaf-4a9b-b761-c8edca48b2b2","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T13:38:25.013054Z","iopub.execute_input":"2021-06-10T13:38:25.013443Z","iopub.status.idle":"2021-06-10T13:38:25.174619Z","shell.execute_reply.started":"2021-06-10T13:38:25.013417Z","shell.execute_reply":"2021-06-10T13:38:25.173945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## How many images consist of how many bounding boxes and how are they distributed?","metadata":{"_uuid":"6c6588cd-510c-480e-86fd-b76a327ed87f","_cell_guid":"c4f2eb4a-69ea-409d-a02a-246cb89c7b20","trusted":true}},{"cell_type":"code","source":"a = train['label'].str.count('opacity').value_counts().sort_index()\n\nplot = a.plot(kind = 'bar')\nplt.title('How many images consist of how many boxes')\nplt.xlabel('Number of boxes in the image', fontsize = 12)\nplt.ylabel('Number of images', fontsize = 12)\nplt.xticks(rotation = 0)\n\nfor bar in plot.patches:\n      plot.annotate(bar.get_height(), \n                   (bar.get_x() + bar.get_width() / 2, \n                    bar.get_height()), ha='center', va='center',\n                   size=14, xytext=(0, 8),\n                   textcoords='offset points')\n        \nfor index in a.index:\n    print(f'{round(a[index]/train_len * 100, 2)}% of images consist of {index} bounding boxes\\n')\nprint('\\n')","metadata":{"_uuid":"63a05819-13ec-4c2b-ac5c-1e530e2a39ca","_cell_guid":"a7c55806-3e93-4162-ad8d-31cfe6f39246","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T13:00:20.843396Z","iopub.execute_input":"2021-06-10T13:00:20.843787Z","iopub.status.idle":"2021-06-10T13:00:20.994642Z","shell.execute_reply.started":"2021-06-10T13:00:20.843762Z","shell.execute_reply":"2021-06-10T13:00:20.993784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### OOps! I thought images consist of 3 boxes at max. But there is an image which consist of 8 boxes.","metadata":{"_uuid":"950bab95-c487-4b0c-b6b1-1c8f328d558a","_cell_guid":"896adc25-c069-4e00-a2b1-631ce1624c8c","trusted":true}},{"cell_type":"markdown","source":"### Since there are only 2 examples where boxes are more than 4. Therefore we'll drop these 2 examples from the dataset.","metadata":{"_uuid":"da7711ca-18f4-433a-b681-cb94529cca4c","_cell_guid":"d35e2a1b-a83f-45aa-b1f6-ad68c6b3fb3e","trusted":true}},{"cell_type":"code","source":"ind_to_drop = train[train['label'].str.count('opacity') > 4].index\ntrain = train.drop(ind_to_drop)\nprint(train.shape)","metadata":{"_uuid":"c90f83f8-0de3-44b3-b378-8466b2f0ba5f","_cell_guid":"3cd82c9d-b9c0-4ebf-b7a9-d19c31a710fd","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:57:03.964183Z","iopub.execute_input":"2021-06-10T12:57:03.964471Z","iopub.status.idle":"2021-06-10T12:57:03.976225Z","shell.execute_reply.started":"2021-06-10T12:57:03.964443Z","shell.execute_reply":"2021-06-10T12:57:03.975588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# code to check whether the label column consist of two types of labels: \n# 1. label with 'none'\n# 2. label with 'opacity'\n\n# train[(train['label'].str.count('opacity') < 1) & (train['label'].str.count('none') != 1 )]","metadata":{"_uuid":"f72b4a75-8c6c-46b7-a4ce-ef1806954629","_cell_guid":"b217b26a-4a99-4e5a-a9cf-b713bc9b7f11","execution":{"iopub.status.busy":"2021-06-10T12:19:09.116631Z","iopub.execute_input":"2021-06-10T12:19:09.116903Z","iopub.status.idle":"2021-06-10T12:19:09.130775Z","shell.execute_reply.started":"2021-06-10T12:19:09.116868Z","shell.execute_reply":"2021-06-10T12:19:09.130013Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## How is abnormality distributed in the dataset?","metadata":{"_uuid":"b27385bb-6619-4352-a8b7-52debd3198cc","_cell_guid":"c918d983-de73-49f3-8086-b8453e281312","trusted":true}},{"cell_type":"code","source":"study_label = ['Negative for Pneumonia', 'Typical Appearance', 'Indeterminate Appearance', 'Atypical Appearance']\nsum_of_abnorm = {}\nfor col in study_label:\n    sum_of_abnorm[col] = train[col].sum(axis = 0)\n    \n\nplot = pd.Series(sum_of_abnorm).plot(kind = 'barh')\n\nfor bar in plot.patches:\n    plot.annotate(str(bar.get_width()) +' ('+ str(round(bar.get_width()/train_len*100, 2))+')' , \n               (bar.get_width() + 120, \n                bar.get_y() + bar.get_height()/2), ha='center', va='center',\n                   size=14, xytext=(30, 0),\n                   textcoords='offset points')","metadata":{"_uuid":"6ecded8b-7eaf-4719-a942-6bed0deb896b","_cell_guid":"3149fa21-b5cb-49cf-aeaf-25ce8a75975b","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:57:07.543248Z","iopub.execute_input":"2021-06-10T12:57:07.543631Z","iopub.status.idle":"2021-06-10T12:57:07.687659Z","shell.execute_reply.started":"2021-06-10T12:57:07.543607Z","shell.execute_reply":"2021-06-10T12:57:07.686806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1. The above graph shows that around 47% of images consist of Typical Appearance. \n### 2. Around 27% of images do not show any abnormality.\n### 3. The dataset is imbalanced!","metadata":{"_uuid":"b0dff56f-7041-4996-abf3-fdfa8096a70f","_cell_guid":"acdfe0aa-a39e-4543-9b3f-d30c1223ff2e","trusted":true}},{"cell_type":"markdown","source":"## Does each image consist of only one label/abnormality?","metadata":{"_uuid":"cb9f157b-0d90-4b4d-9553-158d2181a775","_cell_guid":"7f822a13-1600-4817-b350-32aec04a4c2b","trusted":true}},{"cell_type":"code","source":"subdf = train.loc[:, study_label].astype('object')\nsubdf['sum'] = 0\nsubdf['sum'] = subdf.sum(axis = 1).astype('object')\na = len(subdf[subdf['sum'] != 1.0])\nprint(f'{a} number of times there are more than 1 label for an image\\n')\nsubdf['sum'].plot()\nplt.yticks([0, 1, 2])\nplt.ylabel('No. of labels for an image', fontsize = 12)\nplt.xlabel('Image index', fontsize = 12);","metadata":{"_uuid":"d0c5d409-aaa3-4e51-9cd6-93b3d90b585d","_cell_guid":"f1887f2c-86eb-4621-9442-e28f60751e5e","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:57:13.949355Z","iopub.execute_input":"2021-06-10T12:57:13.949657Z","iopub.status.idle":"2021-06-10T12:57:14.068862Z","shell.execute_reply.started":"2021-06-10T12:57:13.949629Z","shell.execute_reply":"2021-06-10T12:57:14.067957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Each image is just labelled with 1 class. Hence it is a multi-class classification problem at image level.","metadata":{"_uuid":"2352428a-73c0-4920-bc20-6cbcc4ffd672","_cell_guid":"17aed4dd-546e-47ba-9076-a44eff91aa5f","trusted":true}},{"cell_type":"code","source":"a = subdf.describe().T\na['percent'] = a['freq']/a['count'] * 100\nprint(a)","metadata":{"_uuid":"a56fd943-5749-4c5e-bf3c-0d4bc4d2fcb0","_cell_guid":"2931a75f-46c2-42d0-b81e-e7b76ee51ea9","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:57:18.943659Z","iopub.execute_input":"2021-06-10T12:57:18.943982Z","iopub.status.idle":"2021-06-10T12:57:18.975667Z","shell.execute_reply.started":"2021-06-10T12:57:18.943953Z","shell.execute_reply":"2021-06-10T12:57:18.974749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Does no boxes in an image mean no abnormality?","metadata":{"_uuid":"a5a93cf4-949a-4825-9882-778c992c584e","_cell_guid":"e1eee33e-b8e6-4d2d-9dab-0acaf60bdd26","trusted":true}},{"cell_type":"code","source":"none_pneu, none_typical, none_indet, none_atypical = 0, 0, 0, 0\nbox_pneu, box_typical, box_indet, box_atypical = 0, 0, 0, 0\n\nfor index, row in train.iterrows():\n\n    if 'none' in row['label']: # if no bounding box in the image\n        none_pneu += row['Negative for Pneumonia']\n        none_typical += row['Typical Appearance']\n        none_indet += row['Indeterminate Appearance']\n        none_atypical += row['Atypical Appearance']\n        \n    else: # if atleast one box is present in the image\n        box_pneu += row['Negative for Pneumonia']\n        box_typical += row['Typical Appearance']\n        box_indet += row['Indeterminate Appearance']\n        box_atypical += row['Atypical Appearance']\n        \na = {'BBox': ['Absent', 'Present'], 'Negative for Pneumonia': [none_pneu, box_pneu], 'Typical Appearance': [none_typical, box_typical],\n    'Indeterminate Appearance': [none_indet, box_indet], 'Atypical Appearance': [none_atypical, box_atypical]}\n\na = pd.DataFrame(a).set_index('BBox')\n\nplot = a.plot(kind = 'bar')\nplt.xticks(rotation = 0)\nplt.ylabel('No. of images')\n\nfor bar in plot.patches:\n    plot.annotate(bar.get_height(), \n               (bar.get_x() + bar.get_width() / 2, \n                bar.get_height()), ha='center', va='center',\n                   size=14, xytext=(0, 8),\n                   textcoords='offset points')","metadata":{"_uuid":"fcc3607f-1856-46ad-b2ab-a03f55ee4c30","_cell_guid":"9dd917a0-87c4-486a-9d87-6189b3b0414e","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:57:24.473521Z","iopub.execute_input":"2021-06-10T12:57:24.473869Z","iopub.status.idle":"2021-06-10T12:57:25.274280Z","shell.execute_reply.started":"2021-06-10T12:57:24.473841Z","shell.execute_reply":"2021-06-10T12:57:25.273593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### From the above plot:\n### 1. It is not necessary that if there are no bounding boxes in the image, then there is no abnormality. From the plot we can see that, even when bounding boxes are absent there are labels of abnormalities.\n### 2. If there is atleast a single bounding box in the image, then we can say that it consist of some abnormality.","metadata":{"_uuid":"ad30a161-d461-4dba-93f2-eeac2f83312a","_cell_guid":"06908a6a-cc13-44dd-b255-de56af28afea","trusted":true}},{"cell_type":"markdown","source":"## Does each study consist of only one label?","metadata":{"_uuid":"454549de-f231-480b-aaaf-dc6ff45996db","_cell_guid":"fb066054-6bb0-41bb-8636-55a4eb8495fc","trusted":true}},{"cell_type":"code","source":"labels = ['study_id'] + study_label \nsubdf = train[labels]\nprint(subdf.info())","metadata":{"_uuid":"5ac1a9e5-6aa4-4dbb-84bf-4725492e546f","_cell_guid":"09397169-a9ce-4442-95a5-70949bcbe720","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:57:31.149051Z","iopub.execute_input":"2021-06-10T12:57:31.149621Z","iopub.status.idle":"2021-06-10T12:57:31.164896Z","shell.execute_reply.started":"2021-06-10T12:57:31.149590Z","shell.execute_reply":"2021-06-10T12:57:31.163943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame()\nfor col in study_label:\n    group = subdf.groupby('study_id')[col].sum().sort_values(ascending = False)\n    df[col] = group.value_counts()\n\nfig, axes = plt.subplots(2, 2, figsize = (20, 15))\n\nfor col, ax in zip(study_label, axes.ravel()):\n    plot = df[col].plot(kind = 'bar', ax = ax, rot =0, title = col, sharex = True, sharey = True)\n    ax.set_xlabel(f'Number of times the study is positive for abnormality')\n    ax.set_ylabel('Number of studies')\n\n    for bar in plot.patches:\n        plot.annotate(bar.get_height(), \n                   (bar.get_x() + bar.get_width() / 2, \n                    bar.get_height()), ha='center', va='center',\n                       size=14, xytext=(0, 8),\n                       textcoords='offset points')","metadata":{"_uuid":"b9e34387-2cb3-4bf3-946a-a92904ae0dce","_cell_guid":"42ea6c3c-1c75-4d0b-acf6-4f7cc366cedd","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:57:36.759032Z","iopub.execute_input":"2021-06-10T12:57:36.759466Z","iopub.status.idle":"2021-06-10T12:57:37.466668Z","shell.execute_reply.started":"2021-06-10T12:57:36.759437Z","shell.execute_reply":"2021-06-10T12:57:37.465797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Most of the studies are highly imbalanced for abnormalities.","metadata":{"_uuid":"1be7c06a-2f8c-40c5-8408-081993c23bf1","_cell_guid":"4ef381a2-c480-4921-bf68-54287c66edfc","trusted":true}},{"cell_type":"markdown","source":"## What is Specific Character Set in dcm meta data? Is it useful for modelling?","metadata":{"_uuid":"905448e4-a177-4cb3-975e-a0ccb20a05df","_cell_guid":"0aa13df2-378f-4dec-b630-af8537991fa4","trusted":true}},{"cell_type":"markdown","source":"### Found out that it is some kind of encoding for dcm data. Hence it's of no use for modelling. So we'll drop the column from the df.","metadata":{"_uuid":"371eaada-1096-44f4-a69d-f94c44728fb8","_cell_guid":"01c9744d-63d7-4e83-ac5e-403a00a399b7","trusted":true}},{"cell_type":"code","source":"# train['StudyDate'].value_counts()","metadata":{"_uuid":"31248ddb-638e-483a-bf40-f6a0631aa6b6","_cell_guid":"96bfdff2-8858-423c-9446-a5f3aed9098c","jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2021-06-10T12:19:09.223766Z","iopub.execute_input":"2021-06-10T12:19:09.224024Z","iopub.status.idle":"2021-06-10T12:19:09.235294Z","shell.execute_reply.started":"2021-06-10T12:19:09.223994Z","shell.execute_reply":"2021-06-10T12:19:09.234651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train.groupby('StudyDate')[['image_id', 'study_id']].count().sort_values(by = 'image_id', ascending = False).T.plot(kind = 'bar', legend = None, rot = 0);","metadata":{"_uuid":"5af981db-a0cf-4458-99ae-3e4fab9af3cd","_cell_guid":"e28d6abb-8696-4c81-8491-10f843de59c8","jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2021-06-10T12:19:09.236358Z","iopub.execute_input":"2021-06-10T12:19:09.236620Z","iopub.status.idle":"2021-06-10T12:19:09.248053Z","shell.execute_reply.started":"2021-06-10T12:19:09.236593Z","shell.execute_reply":"2021-06-10T12:19:09.246783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# l = ['study_id'] + study_label\n# def func(df):\n#     for col in df.columns:\n#         if col == 'study_id':\n#             df['study_id'] = df['study_id'].count()\n#         else:\n#             df[col] = df[col].sum()\n#     return df\n    \n# train.groupby('StudyDate')[l].apply(lambda df: func(df)).sort_values(by = 'Negative for Pneumonia', ascending = False).tail(10)","metadata":{"_uuid":"f259761c-f13d-406b-ae8b-4fdb795f3029","_cell_guid":"013bfed1-ea5d-4a9e-b1d4-01be5f48df61","jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2021-06-10T12:19:09.249232Z","iopub.execute_input":"2021-06-10T12:19:09.249555Z","iopub.status.idle":"2021-06-10T12:19:09.260189Z","shell.execute_reply.started":"2021-06-10T12:19:09.249496Z","shell.execute_reply":"2021-06-10T12:19:09.259318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(train['AccessionNumber'].nunique())\n# print('Equal to number of studies')","metadata":{"_uuid":"f57642a5-6789-4858-95d6-9eb3fdd7056a","_cell_guid":"8942e089-cbf8-46a5-bd02-75767144631b","jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2021-06-10T12:19:09.261333Z","iopub.execute_input":"2021-06-10T12:19:09.261669Z","iopub.status.idle":"2021-06-10T12:19:09.272175Z","shell.execute_reply.started":"2021-06-10T12:19:09.261637Z","shell.execute_reply":"2021-06-10T12:19:09.271133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(train['Modality'].value_counts(normalize = True))\n\n# plot = train.groupby('Modality')[study_label].sum().plot(kind = 'bar', rot = 0)\n\n# for bar in plot.patches:\n#     plot.annotate(bar.get_height(), \n#                (bar.get_x() + bar.get_width() / 2, \n#                 bar.get_height()), ha='center', va='center',\n#                    size=14, xytext=(0, 8),\n#                    textcoords='offset points')","metadata":{"_uuid":"fdc220f1-2b02-4c41-bff8-a851f9a915b9","_cell_guid":"904fef47-ee72-49ff-802d-31ca60bf7907","jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2021-06-10T12:19:09.273637Z","iopub.execute_input":"2021-06-10T12:19:09.273918Z","iopub.status.idle":"2021-06-10T12:19:09.283877Z","shell.execute_reply.started":"2021-06-10T12:19:09.273891Z","shell.execute_reply":"2021-06-10T12:19:09.282917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train['PatientName'].nunique())","metadata":{"_uuid":"67763a63-226a-4988-b136-f8261755204d","_cell_guid":"60b82b05-ffa6-4060-a3b3-ff45cdd56cb0","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:57:43.714165Z","iopub.execute_input":"2021-06-10T12:57:43.714500Z","iopub.status.idle":"2021-06-10T12:57:43.721618Z","shell.execute_reply.started":"2021-06-10T12:57:43.714471Z","shell.execute_reply":"2021-06-10T12:57:43.720589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot = train.groupby('PatientName')['study_id'].count().value_counts().sort_index().plot(kind = 'bar', rot = 0)\n\nplt.xlabel('Number of studies a patient is part of')\nplt.ylabel('Number of patients')\nfor bar in plot.patches:\n    plot.annotate(bar.get_height(), \n               (bar.get_x() + bar.get_width() / 2, \n                bar.get_height()), ha='center', va='center',\n                   size=14, xytext=(0, 8),\n                   textcoords='offset points')","metadata":{"_uuid":"d61a4543-58b6-40e7-ba4b-57ca91ee0fa8","_cell_guid":"cb1c9542-04c2-4f57-821e-06dda32407ed","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:57:47.508935Z","iopub.execute_input":"2021-06-10T12:57:47.509260Z","iopub.status.idle":"2021-06-10T12:57:47.977000Z","shell.execute_reply.started":"2021-06-10T12:57:47.509233Z","shell.execute_reply":"2021-06-10T12:57:47.976150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### There are patients who are part of 26 studies.","metadata":{"_uuid":"605359b6-58f3-4003-8edd-4b33c4c5d969","_cell_guid":"3c9847a0-0670-41b9-a8c2-1203da3f6c8f","trusted":true}},{"cell_type":"code","source":"print(train['PatientID'].nunique())","metadata":{"_uuid":"3547741a-b76c-497a-8151-32520ae69d0d","_cell_guid":"82414706-766a-4d22-b889-347453d72b7b","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:57:53.348357Z","iopub.execute_input":"2021-06-10T12:57:53.348707Z","iopub.status.idle":"2021-06-10T12:57:53.355589Z","shell.execute_reply.started":"2021-06-10T12:57:53.348679Z","shell.execute_reply":"2021-06-10T12:57:53.354822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train['PatientSex'].value_counts(normalize = True))","metadata":{"_uuid":"b1f8c3ec-0fd6-4bfd-ae76-ae4eed033d1d","_cell_guid":"eb42d1a5-a3e5-4c83-be50-80e864b6d0ee","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:57:56.988001Z","iopub.execute_input":"2021-06-10T12:57:56.988270Z","iopub.status.idle":"2021-06-10T12:57:56.996629Z","shell.execute_reply.started":"2021-06-10T12:57:56.988245Z","shell.execute_reply":"2021-06-10T12:57:56.995730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot = train.groupby('PatientSex')[study_label].sum().plot(kind = 'bar', rot = 0)\nplt.ylabel('Number of Patients')\n\nfor bar in plot.patches:\n    plot.annotate(bar.get_height(), \n               (bar.get_x() + bar.get_width() / 2, \n                bar.get_height()), ha='center', va='center',\n                   size=14, xytext=(0, 8),\n                   textcoords='offset points')","metadata":{"_uuid":"47772761-7297-4def-8df1-a406bea96936","_cell_guid":"43180826-127e-46b1-b342-0cf411251096","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:58:01.633750Z","iopub.execute_input":"2021-06-10T12:58:01.634018Z","iopub.status.idle":"2021-06-10T12:58:01.811633Z","shell.execute_reply.started":"2021-06-10T12:58:01.633993Z","shell.execute_reply":"2021-06-10T12:58:01.811130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Gender seems to be useful, since men tends to have Typical Appearance of Covid-19 more than women in the dataset.","metadata":{"_uuid":"6543aecd-1535-438b-8680-d67cfb03da5d","_cell_guid":"415925c1-c6b2-4f0f-9384-cfa6023fd429","trusted":true}},{"cell_type":"code","source":"def func(df):\n    d = {}\n    for col in df.columns:\n        if col == 'image_id':\n            d['No. of imgs'] = df[col].count()\n        else:\n            d[col] = df[col].sum()\n    d = pd.Series(d)\n    return d\n\ntrain.groupby('BodyPartExamined', dropna = False)[['image_id'] + study_label].apply(func)","metadata":{"_uuid":"093c077f-5fb0-412d-861e-ee4854ee666d","_cell_guid":"1bbea1bf-4df3-4559-bb9e-f5a6d89bb27b","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:58:19.284037Z","iopub.execute_input":"2021-06-10T12:58:19.284567Z","iopub.status.idle":"2021-06-10T12:58:19.317512Z","shell.execute_reply.started":"2021-06-10T12:58:19.284520Z","shell.execute_reply":"2021-06-10T12:58:19.316626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Since the Nan in 'BodyPartExamined' matches with TORAX label, so we'll fill nan values with TORAX","metadata":{"_uuid":"65f4409f-70e2-4725-a97e-550d40e1cf16","_cell_guid":"f06574b2-e5f4-483e-91e0-d84749797fcd","trusted":true}},{"cell_type":"code","source":"train['BodyPartExamined'].fillna('TORAX', inplace = True)","metadata":{"_uuid":"a0a26c71-420b-41e1-bfa4-01f9c3d9e6de","_cell_guid":"e853e58d-b261-4e75-852a-12ce562ad5f5","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:58:25.278892Z","iopub.execute_input":"2021-06-10T12:58:25.279216Z","iopub.status.idle":"2021-06-10T12:58:25.285256Z","shell.execute_reply.started":"2021-06-10T12:58:25.279188Z","shell.execute_reply":"2021-06-10T12:58:25.283803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"index_ = train[train['BodyPartExamined'] == 'SKULL'].index\nindex_ = np.random.choice(index_, 5)\n\nfor ind in index_:\n    plot_bbox(train.loc[ind, 'image_path'], train.loc[ind, 'label'], train.loc[ind, 'BodyPartExamined'])","metadata":{"_uuid":"0320df4c-60f9-4781-bd15-ab8be6cdd53a","_cell_guid":"ab7eb1f9-0514-4585-8bf0-0eaa9227974f","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:58:36.144056Z","iopub.execute_input":"2021-06-10T12:58:36.144339Z","iopub.status.idle":"2021-06-10T12:58:43.118598Z","shell.execute_reply.started":"2021-06-10T12:58:36.144304Z","shell.execute_reply":"2021-06-10T12:58:43.117960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train['PhotometricInterpretation'].value_counts())   #normalize = True","metadata":{"_uuid":"e0a134c4-604a-435d-9f53-0fea2fd37263","_cell_guid":"f581fe3a-68bb-4182-b80d-3eccfb1927e1","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:58:47.928347Z","iopub.execute_input":"2021-06-10T12:58:47.928656Z","iopub.status.idle":"2021-06-10T12:58:47.934364Z","shell.execute_reply.started":"2021-06-10T12:58:47.928630Z","shell.execute_reply":"2021-06-10T12:58:47.933281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def func(df):\n    d = {}\n    for col in df.columns:\n        if col == 'image_id':\n            d['No. of imgs'] = df[col].count()\n        else:\n            d[col] = df[col].sum()\n    d = pd.Series(d)\n    return d\n\na = train.groupby('PhotometricInterpretation')[['image_id'] + study_label].apply(func)\na","metadata":{"_uuid":"17476761-a0da-40e9-943f-545e93989444","_cell_guid":"a6ba4ddc-5ef6-4c58-9266-6ad0838f810a","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:59:02.899518Z","iopub.execute_input":"2021-06-10T12:59:02.899801Z","iopub.status.idle":"2021-06-10T12:59:02.917697Z","shell.execute_reply.started":"2021-06-10T12:59:02.899778Z","shell.execute_reply":"2021-06-10T12:59:02.916780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot = a.plot(kind = 'bar', rot = 0)\nplt.ylabel('No. of images')\n\nfor bar in plot.patches:\n    plot.annotate(bar.get_height(), \n               (bar.get_x() + bar.get_width() / 2, \n                bar.get_height()), ha='center', va='center',\n                   size=14, xytext=(0, 8),\n                   textcoords='offset points')","metadata":{"_uuid":"9410310b-90a8-43dd-9e10-bccb3a0695ad","_cell_guid":"629fcdfd-1c85-488f-8095-e4cfdc43d27b","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:59:09.973183Z","iopub.execute_input":"2021-06-10T12:59:09.973433Z","iopub.status.idle":"2021-06-10T12:59:10.162626Z","shell.execute_reply.started":"2021-06-10T12:59:09.973411Z","shell.execute_reply":"2021-06-10T12:59:10.161757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## We'll drop the columns which seems of no use for modelling.","metadata":{"_uuid":"b2886033-34b7-4780-86bb-1f03719c2366","_cell_guid":"174cbbd9-d819-46ba-9a56-e91de83efa4d","trusted":true}},{"cell_type":"code","source":"train = train.drop(['SpecificCharacterSet', 'SOPClassUID', 'SOPInstanceUID', 'AccessionNumber', 'StudyDate', 'StudyTime',\n                   'Modality', 'PatientID', 'PhotometricInterpretation'], 1)\ntrain.to_csv('./TRAIN.csv', index = False)\nprint('File saved.')","metadata":{"_uuid":"23c56d06-c4cb-446b-a02d-df53a67a100e","_cell_guid":"a46c1b7a-945c-47e4-91d5-31649318035a","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:59:14.949026Z","iopub.execute_input":"2021-06-10T12:59:14.949463Z","iopub.status.idle":"2021-06-10T12:59:15.019902Z","shell.execute_reply.started":"2021-06-10T12:59:14.949434Z","shell.execute_reply":"2021-06-10T12:59:15.019033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.info())","metadata":{"_uuid":"3eddf8c1-d4e2-46b1-a04f-8e8314684126","_cell_guid":"eea17ec2-f33f-4c9e-bbdf-d2eae273056e","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-10T12:59:19.992830Z","iopub.execute_input":"2021-06-10T12:59:19.993100Z","iopub.status.idle":"2021-06-10T12:59:20.008068Z","shell.execute_reply.started":"2021-06-10T12:59:19.993077Z","shell.execute_reply":"2021-06-10T12:59:20.007237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Thank you for going through it. Any improvement or correction please let me know.","metadata":{}},{"cell_type":"markdown","source":"### All the Best for the competition and if you found this notebook useful, upvotes will be appreciated.","metadata":{}}]}