{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<span style=\"color:#F10086;font-size:24Px;\">**<u><b>PROJECT OBJECTIVE:</b></u>  Design a DL based algorithm for detecting pneumonia.</span>**<br/>","metadata":{}},{"cell_type":"markdown","source":"##### <span style=\"color:#3E00FF;font-size:24Px;\">1. Milestone 1:</span>\n> ##### <span style=\"color:#3E00FF;font-size:18Px;\">Input: Context and Dataset</span>\n> ##### <span style=\"color:#3E00FF;font-size:18Px;\">Process:</span>\n>> ##### <span style=\"color:#3E00FF;font-size:18Px;\">Step 1: Import the data. (Milestone Steps)</span>\n>>> ##### <span style=\"color:#F10086;font-size:18Px;\">Step 1: Import the data. (Project Steps)</span>\n>> ##### <span style=\"color:#3E00FF;font-size:18Px;\">Step 2: Map training and testing images to its classes.</span>\n>> ##### <span style=\"color:#3E00FF;font-size:18Px;\">Step 3: Map training and testing images to its annotations.</span>\n>> ##### <span style=\"color:#3E00FF;font-size:18Px;\">Step 4: Preprocessing and Visualisation of different classes.</span>\n>> ##### <span style=\"color:#3E00FF;font-size:18Px;\">Step 5: Display images with bounding box.</span>\n>>> ##### <span style=\"color:#F10086;font-size:18Px;\">Step 2: Exploratory Data Analysis (EDA)</span>\n>>> ##### <span style=\"color:#F10086;font-size:18Px;\">Step 3: Reading Images</span>\n>>> ##### <span style=\"color:#F10086;font-size:18Px;\">Step 3-1: Display images with bounding box</span>\n>> ##### <span style=\"color:#3E00FF;font-size:18Px;\">Step 6: Design, train and test basic CNN models for classification.</span>\n>>> ##### <span style=\"color:#F10086;font-size:18Px;\">Step 4: Design, train and test basic CNN models for classification.</span>\n>> ##### <span style=\"color:#3E00FF;font-size:18Px;\">Step 7: Interim report.</span>\n","metadata":{}},{"cell_type":"markdown","source":"##### <span style=\"color:#3E00FF;font-size:24Px;\">Step 1: Import the data. (Milestone Steps)</span>\n>> ##### <span style=\"color:#F10086;font-size:24Px;\"><u><b>Step1:</b></u> Import the Data</span>","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd \nimport numpy as np\nimport matplotlib\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm_notebook\nfrom matplotlib.patches import Rectangle\nimport seaborn as sns\nimport pydicom ","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:55:45.445169Z","iopub.execute_input":"2022-09-17T09:55:45.446032Z","iopub.status.idle":"2022-09-17T09:55:46.103706Z","shell.execute_reply.started":"2022-09-17T09:55:45.445945Z","shell.execute_reply":"2022-09-17T09:55:46.102789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"color:#F10086;font-size:18Px;\">**1. The dataset size is huge, so we are using kaggle notebook for implementation.<br/>2. We can directly access the dataset and train the model on free GPU.</span>**","metadata":{}},{"cell_type":"code","source":"os.listdir(\"/kaggle/input/rsna-pneumonia-detection-challenge/\")","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:55:48.161578Z","iopub.execute_input":"2022-09-17T09:55:48.161977Z","iopub.status.idle":"2022-09-17T09:55:48.172315Z","shell.execute_reply.started":"2022-09-17T09:55:48.161943Z","shell.execute_reply":"2022-09-17T09:55:48.171197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = pd.read_csv(\"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv\")\ntrain_dataset.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:55:50.121671Z","iopub.execute_input":"2022-09-17T09:55:50.122167Z","iopub.status.idle":"2022-09-17T09:55:50.220946Z","shell.execute_reply.started":"2022-09-17T09:55:50.122126Z","shell.execute_reply":"2022-09-17T09:55:50.219987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_classfier = pd.read_csv(\"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_detailed_class_info.csv\")\ntrain_data_classfier.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:55:57.582144Z","iopub.execute_input":"2022-09-17T09:55:57.583229Z","iopub.status.idle":"2022-09-17T09:55:57.663511Z","shell.execute_reply.started":"2022-09-17T09:55:57.583184Z","shell.execute_reply":"2022-09-17T09:55:57.662492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Train Labels dataframe has {train_dataset.shape[0]} rows and {train_dataset.shape[1]} columns')\nprint(f'Class info dataframe has {train_data_classfier.shape[0]} rows and {train_data_classfier.shape[1]} columns')\nprint('Number of duplicates in patientID in train labels dataframe: {}'.format(len(train_dataset) - (train_dataset['patientId'].nunique())))\nprint('Number of duplicates in patientID in class info dataframe: {}'.format(len(train_data_classfier) - (train_data_classfier['patientId'].nunique())))","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:55:59.809963Z","iopub.execute_input":"2022-09-17T09:55:59.810322Z","iopub.status.idle":"2022-09-17T09:55:59.836729Z","shell.execute_reply.started":"2022-09-17T09:55:59.810286Z","shell.execute_reply":"2022-09-17T09:55:59.835506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset.info()","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:31:46.013249Z","iopub.execute_input":"2022-09-17T09:31:46.013730Z","iopub.status.idle":"2022-09-17T09:31:46.034829Z","shell.execute_reply.started":"2022-09-17T09:31:46.013695Z","shell.execute_reply":"2022-09-17T09:31:46.033416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Numbers of nulls in bounding boxes columns are equal to the 0s we have in Target column')\nprint('Checking nulls in bounding boxes columns: {}'.format(train_dataset[['x', 'y', 'width', 'height']].isnull().sum().to_dict())) \nprint('Checking value counts for the targets: {}'.format(train_dataset['Target'].value_counts().to_dict()))","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:31:48.440561Z","iopub.execute_input":"2022-09-17T09:31:48.440938Z","iopub.status.idle":"2022-09-17T09:31:48.452702Z","shell.execute_reply.started":"2022-09-17T09:31:48.440904Z","shell.execute_reply":"2022-09-17T09:31:48.451523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### <span style=\"color:#3E00FF;font-size:24Px;\">Step 2: Map training and testing images to its classes.</span>\n##### <span style=\"color:#3E00FF;font-size:24Px;\">Step 3: Map training and testing images to its annotations.</span>\n##### <span style=\"color:#3E00FF;font-size:24Px;\">Step 4: Preprocessing and Visualisation of different classes.</span>\n##### <span style=\"color:#3E00FF;font-size:24Px;\">Step 5: Display images with bounding box.</span>\n>> ##### <span style=\"color:#F10086;font-size:24Px;\"><u><b>Step2:</b></u> Exploratory Data Analysis (EDA)</span>","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"<span style=\"color:#F10086;font-size:24Px;\"><b><u>1. Checking the class distribution.</u></b></span>","metadata":{}},{"cell_type":"code","source":"def classwise_distribution(data,column_name):\n    class_counts = data[column_name].value_counts()\n    sample_length = len(class_counts)\n    \n    print(\"Feature: {}\".format(column_name))\n    for i in range(sample_length):\n        label = class_counts.index[i]\n        count = class_counts.values[i]\n        percent = int((count/len(data))*10000)/100\n        print(\"{:<30s}: {} or {}%\".format(label, count, percent))","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:31:54.361079Z","iopub.execute_input":"2022-09-17T09:31:54.361485Z","iopub.status.idle":"2022-09-17T09:31:54.367005Z","shell.execute_reply.started":"2022-09-17T09:31:54.361453Z","shell.execute_reply":"2022-09-17T09:31:54.366220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classwise_distribution(train_data_classfier, 'class')","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:32:00.826889Z","iopub.execute_input":"2022-09-17T09:32:00.827266Z","iopub.status.idle":"2022-09-17T09:32:00.834380Z","shell.execute_reply.started":"2022-09-17T09:32:00.827240Z","shell.execute_reply":"2022-09-17T09:32:00.833646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(train_data_classfier[\"class\"].values)\nplt.title(\"Class Label Distribution\")\nplt.xlabel(\"Training Class Label\")\nplt.ylabel(\"Count\")","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:32:03.538090Z","iopub.execute_input":"2022-09-17T09:32:03.538491Z","iopub.status.idle":"2022-09-17T09:32:03.703040Z","shell.execute_reply.started":"2022-09-17T09:32:03.538460Z","shell.execute_reply":"2022-09-17T09:32:03.702234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(10,6))\ng = (train_data_classfier['class'].value_counts().sort_index(ascending=False).plot(kind='pie', autopct='%.0f%%', colors=['deepskyblue', 'lightpink', 'salmon'], startangle=100, title='Distribution of Class', fontsize=12).set_ylabel(''))","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:32:12.081308Z","iopub.execute_input":"2022-09-17T09:32:12.082160Z","iopub.status.idle":"2022-09-17T09:32:12.179006Z","shell.execute_reply.started":"2022-09-17T09:32:12.082126Z","shell.execute_reply":"2022-09-17T09:32:12.178128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"color:#F10086;font-size:24Px;\"><b><u>2. Checking the number of bounding boxes for each unique patient.</u></b></span>","metadata":{}},{"cell_type":"code","source":"print('Grouping by patient IDs and checking the number of bounding boxes for each unique patient ID\\n')\n\nboundingboxes = train_dataset.groupby('patientId').size().to_frame('number_of_boxes').reset_index()\ntrain_dataset = train_dataset.merge(boundingboxes, on='patientId', how='left')\nprint('Number of unique patient IDs in the dataset: {}'.format(len(boundingboxes)))\nprint('Number of patient IDs per bounding boxes in the dataset')\n(boundingboxes.groupby('number_of_boxes').size().to_frame('number_of_patientIDs_per_boxes').reset_index().set_index('number_of_boxes').sort_values(by='number_of_boxes'))","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:32:15.530713Z","iopub.execute_input":"2022-09-17T09:32:15.532016Z","iopub.status.idle":"2022-09-17T09:32:15.578786Z","shell.execute_reply.started":"2022-09-17T09:32:15.531981Z","shell.execute_reply":"2022-09-17T09:32:15.577614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_train_dataset = pd.concat([train_dataset, train_data_classfier['class']], axis = 1)\nnew_train_dataset.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:56:08.448559Z","iopub.execute_input":"2022-09-17T09:56:08.448943Z","iopub.status.idle":"2022-09-17T09:56:08.467435Z","shell.execute_reply.started":"2022-09-17T09:56:08.448909Z","shell.execute_reply":"2022-09-17T09:56:08.466469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Shape of the dataset after number of boxes column: {}'.format(new_train_dataset.shape))","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:32:20.443839Z","iopub.execute_input":"2022-09-17T09:32:20.444224Z","iopub.status.idle":"2022-09-17T09:32:20.451507Z","shell.execute_reply.started":"2022-09-17T09:32:20.444193Z","shell.execute_reply":"2022-09-17T09:32:20.450212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<span style=\"color:#F10086;font-size:24Px;\">**<u><b>Observation:</b></u></span>**<br/>\n\n<span style=\"color:#F10086;font-size:18Px;\">**1. Training data contains patientIds and bounding boxes as x, y, width and height.</span>**<br/>\n<span style=\"color:#F10086;font-size:18Px;\">**2. There are multiple records for patients. Number of duplicates in patientID = 3543.</span>**<br/>\n<span style=\"color:#F10086;font-size:18Px;\">**3. There is also a binary target column i.e. Target indicating presence or absence of pneumonia.</span>**<br/>\n<span style=\"color:#F10086;font-size:18Px;\">**4. Class label contains: No Lung Opacity/Not Normal, Normal and Lung Opacity.</span>**<br/>\n<span style=\"color:#F10086;font-size:18Px;\">**5. Chest examination with Target=1, presence of pneumonia are associated with the Lung Opacity class.</span>**<br/>\n<span style=\"color:#F10086;font-size:18Px;\">**6. Chest examination with Target=0, absence of pneumonia are associated with either as Normal or No Lung Opacity/Not Normal class.</span>**<br/>\n<span style=\"color:#F10086;font-size:18Px;\">**7. About 23286 patientIds has 1 bounding boxes, while 13 patients have 4 bounding boxes.</span>**","metadata":{}},{"cell_type":"markdown","source":"<span style=\"color:#F10086;font-size:24Px;\"><b><u>3. Plotting number of class within target value.</u></b></span>","metadata":{}},{"cell_type":"code","source":"ig, ax = plt.subplots(nrows=1,figsize=(8,6))\n\ntmp = new_train_dataset.groupby('Target')['class'].value_counts()\ndf = pd.DataFrame(data={'Exams':tmp.values},index=tmp.index).reset_index()\n\nsns.barplot(ax=ax, x='Target', y='Exams', hue='class', data=df, palette=\"Blues\")  \nplt.title(\"Distribution of Number of Class within Target Value\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:32:23.419891Z","iopub.execute_input":"2022-09-17T09:32:23.420283Z","iopub.status.idle":"2022-09-17T09:32:23.611487Z","shell.execute_reply.started":"2022-09-17T09:32:23.420252Z","shell.execute_reply":"2022-09-17T09:32:23.610564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"color:#F10086;font-size:18Px;\"><b>Target 1</b> represents the <b>lung opicity</b> class.</span><br/>\n<span style=\"color:#F10086;font-size:18Px;\"><b>Target 0</b> represents the <b>No Lung Opacity/Not Normal</b> and <b>Normal</b> class.</span>","metadata":{}},{"cell_type":"markdown","source":"<span style=\"color:#F10086;font-size:24Px;\">**<u><b>4. Feature extraction from the dicom image files:</b></u></span>**\n\n<span style=\"color:#F10086;font-size:18Px;\">**1. Exploring modality.</span>**<br/>\n<span style=\"color:#F10086;font-size:18Px;\">**2. Exploring Body Part Examined.</span>**<br/>\n<span style=\"color:#F10086;font-size:18Px;\">**3. Exploring different view positions in the dataset.</span>**<br/>\n<span style=\"color:#F10086;font-size:18Px;\">**4. To understand distribution of age for the presence and absence of lung opacity/Pneumonia.</span>**<br/>\n<span style=\"color:#F10086;font-size:18Px;\">**5. To understand distribution of gender for the presence and absence of lung opacity/Pneumonia.</span>**","metadata":{}},{"cell_type":"code","source":"TRAIN_IMAGES = \"../input/rsna-pneumonia-detection-challenge/stage_2_train_images/\"\n\nsample_patient_id = new_train_dataset['patientId'][0]\ndcm_file = TRAIN_IMAGES + '{}.dcm'.format(sample_patient_id)\ndcm_data = pydicom.read_file(dcm_file)\n\nprint(dcm_data)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:32:27.205479Z","iopub.execute_input":"2022-09-17T09:32:27.205858Z","iopub.status.idle":"2022-09-17T09:32:27.224845Z","shell.execute_reply.started":"2022-09-17T09:32:27.205828Z","shell.execute_reply":"2022-09-17T09:32:27.224091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"color:#F10086;font-size:18Px;\">1. From the above sample, After reading the dicom file, which contains information that can be used for further analysis such as Patient's Sex, Age, Body Part Examined - **Chest**, View Position - **Anterior-Posterior (AP) or Posterior-Anterior (PA)** and Modality - **Computed Radiography (CR)**.</span><br/>\n<span style=\"color:#F10086;font-size:18Px;\">**2. Size of this image is 1024 x 1024 (Rows x Columns).</span>**","metadata":{}},{"cell_type":"code","source":"from tqdm import tqdm, notebook\n\nvars = ['Modality', 'PatientAge', 'PatientSex', 'BodyPartExamined', 'ViewPosition']\ndata_path = TRAIN_IMAGES\ndef process_dicom_data(data_df, TRAIN_IMAGES):\n    for var in vars:\n        data_df[var] = None\n    image_names = os.listdir(TRAIN_IMAGES)\n    for i, img_name in tqdm(enumerate(image_names)):\n        imagePath = os.path.join(TRAIN_IMAGES,img_name)\n        data_row_img_data = pydicom.dcmread(imagePath)\n        idx = (data_df['patientId']==data_row_img_data.PatientID)\n        data_df.loc[idx,'Modality'] = data_row_img_data.Modality\n        data_df.loc[idx,'PatientAge'] = pd.to_numeric(data_row_img_data.PatientAge)\n        data_df.loc[idx,'PatientSex'] = data_row_img_data.PatientSex\n        data_df.loc[idx,'BodyPartExamined'] = data_row_img_data.BodyPartExamined\n        data_df.loc[idx,'ViewPosition'] = data_row_img_data.ViewPosition","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:32:31.838582Z","iopub.execute_input":"2022-09-17T09:32:31.838958Z","iopub.status.idle":"2022-09-17T09:32:31.848104Z","shell.execute_reply.started":"2022-09-17T09:32:31.838926Z","shell.execute_reply":"2022-09-17T09:32:31.846998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"process_dicom_data(new_train_dataset,TRAIN_IMAGES)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:32:34.845365Z","iopub.execute_input":"2022-09-17T09:32:34.846298Z","iopub.status.idle":"2022-09-17T09:36:47.168842Z","shell.execute_reply.started":"2022-09-17T09:32:34.846263Z","shell.execute_reply":"2022-09-17T09:36:47.167473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"color:#F10086;font-size:18Px;\">**4.1. Exploring modality.</span>**","metadata":{}},{"cell_type":"code","source":"print(\"Modality: train:\",new_train_dataset['Modality'].unique())","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:36:47.182479Z","iopub.execute_input":"2022-09-17T09:36:47.182807Z","iopub.status.idle":"2022-09-17T09:36:47.199660Z","shell.execute_reply.started":"2022-09-17T09:36:47.182783Z","shell.execute_reply":"2022-09-17T09:36:47.198928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"color:#F10086;font-size:18Px;\">**4.2. Exploring Body Part Examined.</span>**","metadata":{}},{"cell_type":"code","source":"print(\"Body Part Examined: train:\",new_train_dataset['BodyPartExamined'].unique())","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:36:47.201518Z","iopub.execute_input":"2022-09-17T09:36:47.202299Z","iopub.status.idle":"2022-09-17T09:36:47.213464Z","shell.execute_reply.started":"2022-09-17T09:36:47.202271Z","shell.execute_reply":"2022-09-17T09:36:47.212151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"color:#F10086;font-size:18Px;\">**4.3. Exploring different view positions in the dataset.</span>**","metadata":{}},{"cell_type":"code","source":"print(\"View Position: train:\",new_train_dataset['ViewPosition'].unique())","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:36:47.215071Z","iopub.execute_input":"2022-09-17T09:36:47.216170Z","iopub.status.idle":"2022-09-17T09:36:47.231042Z","shell.execute_reply.started":"2022-09-17T09:36:47.216128Z","shell.execute_reply":"2022-09-17T09:36:47.230239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Distribution of `ViewPosition')\nfig = plt.figure(figsize = (10, 6))\nax = fig.add_subplot(121)\ng = (new_train_dataset['ViewPosition'].value_counts().plot(kind='pie', autopct='%.0f%%',startangle=90,title='Distribution of ViewPosition\\nOverall', fontsize=12).set_ylabel(''))\nax = fig.add_subplot(122)\ng = (new_train_dataset.loc[new_train_dataset['Target']==1, 'ViewPosition'].value_counts().sort_index(ascending=False).plot(kind='pie', autopct='%.0f%%',startangle=90, counterclock=False, title='Distribution of ViewPosition\\nPresence of Pneumonia',fontsize=12).set_ylabel(''))","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:36:47.232144Z","iopub.execute_input":"2022-09-17T09:36:47.233117Z","iopub.status.idle":"2022-09-17T09:36:47.397049Z","shell.execute_reply.started":"2022-09-17T09:36:47.232943Z","shell.execute_reply":"2022-09-17T09:36:47.396080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"color:#F10086;font-size:18Px;\">**4.4. To understand distribution of age for the presence and absence of lung opacity/Pneumonia.</span>**","metadata":{}},{"cell_type":"code","source":"fig, (ax1,ax2) =plt.subplots(1,2,figsize=(10,6))\n\nsns.histplot(new_train_dataset['PatientAge'], kde=True, ax=ax1)\nax1.set_title('Distribution of PatientAge\\nOverall')\n\nsns.histplot(new_train_dataset.loc[new_train_dataset['Target']==1, 'PatientAge'], kde=True, ax=ax2)\nax2.set_title('Distribution of PatientAge\\nPresence of Pneumonia')","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:36:47.398313Z","iopub.execute_input":"2022-09-17T09:36:47.399072Z","iopub.status.idle":"2022-09-17T09:36:48.327040Z","shell.execute_reply.started":"2022-09-17T09:36:47.399042Z","shell.execute_reply":"2022-09-17T09:36:48.326143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"color:#F10086;font-size:18Px;\"><b>4.4.1 Outlier - Detection and Removing.</b></span>","metadata":{}},{"cell_type":"code","source":"print('Minimum `PatientAge` in the training dataset: {}'.format(new_train_dataset['PatientAge'].min()))\nprint('Maximum `PatientAge` in the training dataset: {}'.format(new_train_dataset['PatientAge'].max()))\nprint('25th Percentile of `PatientAge` in the training dataset: {}'.format(new_train_dataset['PatientAge'].quantile(0.25)))\nprint('50th Percentile of `PatientAge` in the training dataset: {}'.format(new_train_dataset['PatientAge'].quantile(0.50)))\nprint('75th Percentile of `PatientAge` in the training dataset: {}'.format(new_train_dataset['PatientAge'].quantile(0.75)))\nprint('`PatientAge` in upper whisker for box plot: {}'.format(new_train_dataset['PatientAge'].quantile(0.75) + (new_train_dataset['PatientAge'].quantile(0.75) - new_train_dataset['PatientAge'].quantile(0.25))))\nprint()\nfig = plt.figure(figsize=(10, 6))\nax = sns.boxplot(data=new_train_dataset['PatientAge'], orient='h').set_title('PatientAge with outliers')","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:36:48.328127Z","iopub.execute_input":"2022-09-17T09:36:48.328417Z","iopub.status.idle":"2022-09-17T09:36:48.595433Z","shell.execute_reply.started":"2022-09-17T09:36:48.328390Z","shell.execute_reply":"2022-09-17T09:36:48.594652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Setting the upper threshold as 100 for age and removing outliers')\nnew_train_dataset['PatientAge'] = new_train_dataset['PatientAge'].clip(new_train_dataset['PatientAge'].min(), 96)\n\nfig = plt.figure(figsize=(10, 6))\nax = sns.boxplot(data=new_train_dataset['PatientAge'], orient='h').set_title('PatientAge after removing outliers')\n","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:36:48.596608Z","iopub.execute_input":"2022-09-17T09:36:48.597081Z","iopub.status.idle":"2022-09-17T09:36:48.742466Z","shell.execute_reply.started":"2022-09-17T09:36:48.597048Z","shell.execute_reply":"2022-09-17T09:36:48.740879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax1,ax2) =plt.subplots(1,2,figsize=(10,6))\n\nsns.histplot(new_train_dataset['PatientAge'], kde=True, ax=ax1)\nax1.set_title('Distribution of PatientAge\\nOverall\\nafter removing outliers')\n\nsns.histplot(new_train_dataset.loc[new_train_dataset['Target']==1, 'PatientAge'], kde=True, ax=ax2)\nax2.set_title('Distribution of PatientAge\\nPresence of Pneumonia\\nafter removing outliers')","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:36:48.745505Z","iopub.execute_input":"2022-09-17T09:36:48.745787Z","iopub.status.idle":"2022-09-17T09:36:49.572475Z","shell.execute_reply.started":"2022-09-17T09:36:48.745763Z","shell.execute_reply":"2022-09-17T09:36:49.571475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Creating Age Binning')\nnew_train_dataset['AgeBins']=pd.cut(new_train_dataset['PatientAge'], bins=5, precision=0, labels=['<=20', '<=40', '<=60', '<=80', '<=100'])\nnew_train_dataset['AgeBins'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:36:49.573897Z","iopub.execute_input":"2022-09-17T09:36:49.574401Z","iopub.status.idle":"2022-09-17T09:36:49.746619Z","shell.execute_reply.started":"2022-09-17T09:36:49.574373Z","shell.execute_reply":"2022-09-17T09:36:49.745076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(pd.concat([new_train_dataset['AgeBins'].value_counts().sort_index().rename('Counts of Age Bins - Overall'), new_train_dataset.loc[new_train_dataset['Target'] == 1, 'AgeBins'].value_counts().sort_index().rename('Counts of Age Bins - Target=1')], axis=1))\n\nf, (ax1, ax2) = plt.subplots(1, 2, figsize = (10, 6))\ng = sns.countplot(x=new_train_dataset['AgeBins'], ax=ax1).set_title('Count Plot of Age Bins\\nOverall')\ng = sns.countplot(x=new_train_dataset.loc[new_train_dataset['Target'] == 1, 'AgeBins'], ax=ax2).set_title('Count Plot of Age Bins\\nPresence of Pneumonia')\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:36:49.747824Z","iopub.execute_input":"2022-09-17T09:36:49.748230Z","iopub.status.idle":"2022-09-17T09:36:50.049148Z","shell.execute_reply.started":"2022-09-17T09:36:49.748166Z","shell.execute_reply":"2022-09-17T09:36:50.048386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"color:#F10086;font-size:18Px;\">**4.5. To understand distribution of gender for the presence and absence of lung opacity/Pneumonia.</span>**","metadata":{}},{"cell_type":"code","source":"print('Distribution of age for patients with Pneumonia Evidence, by Gender')\ndisplay(pd.concat([new_train_dataset['PatientSex'].value_counts(normalize=True).round(2).sort_values().rename('% Gender, Overall'), new_train_dataset.loc[(new_train_dataset['Target'] == 1), 'PatientSex'].value_counts(normalize=True).round(2).sort_index().rename('% Gender, Target=1')], axis=1))\n\nf, ((ax1, ax2), (ax3, ax4)) = plt.subplots(2, 2, figsize = (14, 14))\ng = sns.histplot(new_train_dataset.loc[(new_train_dataset['Target'] == 1) & (new_train_dataset['PatientSex'] == 'M'), 'PatientAge'], kde=True, ax=ax1)\nax1.set_title('Distribution of Age for Male\\nPresence of Pneumonia')\n\ng = sns.histplot(new_train_dataset.loc[(new_train_dataset['Target'] == 1) & (new_train_dataset['PatientSex'] == 'F'), 'PatientAge'], kde=True, ax = ax2)\nax2.set_title('Distribution of Age for Female\\nPresence of Pneumonia')\n\ng = sns.countplot(x=new_train_dataset['PatientSex'], ax=ax3, palette='Set2')\nax3.set_title('Count Plot of Gender\\nOverall')\n\ng = sns.countplot(x=new_train_dataset.loc[(new_train_dataset['Target'] == 1), 'PatientSex'], ax=ax4, palette='Set2')\nax4.set_title('Count Plot of Gender\\nPresence of Pneumonia')\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:40:50.874951Z","iopub.execute_input":"2022-09-17T09:40:50.875371Z","iopub.status.idle":"2022-09-17T09:40:51.588638Z","shell.execute_reply.started":"2022-09-17T09:40:50.875338Z","shell.execute_reply":"2022-09-17T09:40:51.587678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" <span style=\"color:#F10086;font-size:24Px;\"><b><u>Step-3: Reading Images.</u></b></span>","metadata":{}},{"cell_type":"code","source":"sampleImage0 = \"../input/rsna-pneumonia-detection-challenge/stage_2_train_images/001031d9-f904-4a23-b3e5-2c088acd19c6.dcm\"\nimg_arr = pydicom.read_file(sampleImage0).pixel_array\nprint(\"Image shape :\",img_arr.shape)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:40:55.123147Z","iopub.execute_input":"2022-09-17T09:40:55.123550Z","iopub.status.idle":"2022-09-17T09:40:55.140167Z","shell.execute_reply.started":"2022-09-17T09:40:55.123520Z","shell.execute_reply":"2022-09-17T09:40:55.139337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.style.use('ggplot')\nplt.hist(img_arr.flatten())\nplt.title(\"Distributon of Chest Xray Image Pixel value\")\nplt.xlabel(\"Pixel Value\")\nplt.ylabel(\"Pixel Count\")","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:41:03.610873Z","iopub.execute_input":"2022-09-17T09:41:03.611337Z","iopub.status.idle":"2022-09-17T09:41:03.779377Z","shell.execute_reply.started":"2022-09-17T09:41:03.611302Z","shell.execute_reply":"2022-09-17T09:41:03.778614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"normalized = img_arr/img_arr.max()","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:41:06.673323Z","iopub.execute_input":"2022-09-17T09:41:06.674008Z","iopub.status.idle":"2022-09-17T09:41:06.680971Z","shell.execute_reply.started":"2022-09-17T09:41:06.673965Z","shell.execute_reply":"2022-09-17T09:41:06.679645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.style.use('ggplot')\nplt.hist(normalized.flatten())\nplt.title(\"Distributon of Normalized Image Pixel value\")\nplt.xlabel(\"Pixel Value\")\nplt.ylabel(\"Pixel Count\")","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:41:08.593507Z","iopub.execute_input":"2022-09-17T09:41:08.593856Z","iopub.status.idle":"2022-09-17T09:41:08.768450Z","shell.execute_reply.started":"2022-09-17T09:41:08.593828Z","shell.execute_reply":"2022-09-17T09:41:08.766937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# read the image and preprocess the image\n\ndef read_Image(path):\n    img_arr = pydicom.read_file(path).pixel_array\n    normalized = img_arr/img_arr.max()\n    return np.array(normalized)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:58:16.051728Z","iopub.execute_input":"2022-09-17T09:58:16.052432Z","iopub.status.idle":"2022-09-17T09:58:16.057993Z","shell.execute_reply.started":"2022-09-17T09:58:16.052395Z","shell.execute_reply":"2022-09-17T09:58:16.056876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_train_dataset[\"patientId\"].iloc[10]","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:41:14.251055Z","iopub.execute_input":"2022-09-17T09:41:14.251438Z","iopub.status.idle":"2022-09-17T09:41:14.258214Z","shell.execute_reply.started":"2022-09-17T09:41:14.251410Z","shell.execute_reply":"2022-09-17T09:41:14.256916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob, pylab\n\ndef show_random_images(new_train_dataset):\n    path = \"../input/rsna-pneumonia-detection-challenge/stage_2_train_images/\"\n    w = 10\n    h = 10\n    fig = plt.figure(figsize=(28, 36))\n\n    ax = []\n\n    columns = 3; rows = 3\n    for i in range(1,columns*rows +1):\n        int_val = np.random.randint(10,10000)\n        image0 = read_Image(path+new_train_dataset[\"patientId\"].iloc[int_val]+\".dcm\")\n        ax.append(fig.add_subplot(rows, columns, i))\n        ax[-1].set_title('Patient ID: {}\\nTarget: {}\\nClass: {}\\nWindow: {}:{}:{}:{}'.format(\n                    new_train_dataset['patientId'].iloc[int_val],new_train_dataset['Target'].iloc[int_val], new_train_dataset['class'].iloc[int_val], \n                    new_train_dataset['x'].iloc[int_val],new_train_dataset['y'].iloc[int_val],new_train_dataset['width'].iloc[int_val],new_train_dataset['height'].iloc[int_val]))\n        plt.imshow(image0)\n\n    plt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:41:20.513845Z","iopub.execute_input":"2022-09-17T09:41:20.514220Z","iopub.status.idle":"2022-09-17T09:41:20.522968Z","shell.execute_reply.started":"2022-09-17T09:41:20.514164Z","shell.execute_reply":"2022-09-17T09:41:20.521944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_random_images(new_train_dataset)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:41:24.695259Z","iopub.execute_input":"2022-09-17T09:41:24.696010Z","iopub.status.idle":"2022-09-17T09:41:27.421906Z","shell.execute_reply.started":"2022-09-17T09:41:24.695975Z","shell.execute_reply":"2022-09-17T09:41:27.421163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" <span style=\"color:#F10086;font-size:24Px;\"><b><u>3.1: Display images with bounding box.</u></b></span>","metadata":{}},{"cell_type":"code","source":"def Displayimages_with_bounding_box(data):\n    path = \"../input/rsna-pneumonia-detection-challenge/stage_2_train_images/\"\n    img_data = list(data.T.to_dict().values())\n    f, ax = plt.subplots(3,3, figsize=(16,18))\n    for i,data_row in enumerate(img_data):\n        patientImage = data_row['patientId']+'.dcm'\n        data_row_img_data = read_Image(path+patientImage)\n\n        data_row_img = pydicom.dcmread(path+patientImage)\n        ax[i//3, i%3].imshow(data_row_img.pixel_array, cmap=plt.cm.bone) \n        ax[i//3, i%3].axis('off')\n        ax[i//3, i%3].set_title('ID: {} Target: {}\\nClass: {}'.format(\n                data_row['patientId'],data_row['Target'], data_row['class']))\n        rows = new_train_dataset[new_train_dataset['patientId']==data_row['patientId']]\n        box_data = list(rows.T.to_dict().values())\n        for j, row in enumerate(box_data):\n            ax[i//3, i%3].add_patch(Rectangle(xy=(row['x'], row['y']),\n                        width=row['width'],height=row['height'], \n                        color=\"red\",alpha = 0.2))   \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:41:59.116290Z","iopub.execute_input":"2022-09-17T09:41:59.117623Z","iopub.status.idle":"2022-09-17T09:41:59.127265Z","shell.execute_reply.started":"2022-09-17T09:41:59.117559Z","shell.execute_reply":"2022-09-17T09:41:59.126240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Displayimages_with_bounding_box(new_train_dataset[new_train_dataset['Target']==1].sample(9))\n","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:42:02.007158Z","iopub.execute_input":"2022-09-17T09:42:02.008107Z","iopub.status.idle":"2022-09-17T09:42:03.614761Z","shell.execute_reply.started":"2022-09-17T09:42:02.008062Z","shell.execute_reply":"2022-09-17T09:42:03.613870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### <span style=\"color:#3E00FF;font-size:24Px;\">Step 6: Design, train and test basic CNN models for classification.</span>\n>> ##### <span style=\"color:#F10086;font-size:24Px;\">Step 4: Design, train and test basic CNN models for classification.</span>\n","metadata":{}},{"cell_type":"code","source":"IMAGE_WIDTH = IMAGE_HEIGHT = 256\nCHANNEL = 3","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:57:01.191728Z","iopub.execute_input":"2022-09-17T09:57:01.192406Z","iopub.status.idle":"2022-09-17T09:57:01.197332Z","shell.execute_reply.started":"2022-09-17T09:57:01.192371Z","shell.execute_reply":"2022-09-17T09:57:01.195978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n\nclass simpleDensenet():\n    def __init__(self,epochs,batch_size,lr=0.001):\n        self.epochs = epochs\n        self.batch_size = batch_size\n        self.lr = lr\n        self.dense_depth= 2\n        self.start_neurons = 16\n        \n    def convolutionBlock(self,x, filters, size, strides=(1,1), padding='same'):\n        x = tf.keras.layers.Conv2D(filters, size, strides=strides, padding=padding)(x)\n        x = tf.keras.layers.BatchNormalization()(x)\n        x = tf.keras.layers.Activation('elu')(x)\n        return x\n    \n    def denseBlock(self,blockInput, num_filters=16, batch_activate = False):\n        list_Conv = [blockInput]\n        pas=self.convolutionBlock(blockInput, num_filters,size=(3,3),strides=(1,1))\n        \n        for i in range(2 , self.dense_depth+1):\n            list_Conv.append(pas)\n            out =  tf.keras.layers.Concatenate(axis = 3)(list_Conv) # conctenated out put\n            pas = self.convolutionBlock(out, num_filters,size=(3,3),strides=(1,1))\n\n            list_Conv.append(pas)\n            \n        out = tf.keras.layers.Concatenate(axis =3)(list_Conv)\n        feature = self.convolutionBlock(out, num_filters,size=(3,3),strides=(1,1))\n    \n        return feature\n    \n\n    def build(self,inputShape = (IMAGE_WIDTH, IMAGE_HEIGHT,CHANNEL),number_classes=None):\n        \n        input_layer = tf.keras.layers.Input(inputShape)\n        conv1 = self.convolutionBlock(input_layer,self.start_neurons,(3,3)) \n        dense0 = self.denseBlock(conv1,self.start_neurons)\n        mxpool1 = tf.keras.layers.MaxPooling2D((2, 2))(dense0)\n\n        conv2 = self.convolutionBlock(mxpool1,self.start_neurons*2,(3,3))  \n        dense1 = self.denseBlock(conv2,self.start_neurons*2) \n        mxpool2 = tf.keras.layers.MaxPooling2D((2, 2))(dense1)\n\n        conv3 = self.convolutionBlock(mxpool2,self.start_neurons*4,(3,3))\n        dense3 = self.denseBlock(conv3,self.start_neurons*4) \n        mxpool3 = tf.keras.layers.MaxPooling2D((2, 2))(dense3)\n\n        conv4= self.convolutionBlock(mxpool3,self.start_neurons*4,(3,3))\n        dense4 = self.denseBlock(conv4,self.start_neurons*4) \n        mxpool4 = tf.keras.layers.MaxPooling2D((2, 2))(dense4)\n\n        conv5= self.convolutionBlock(mxpool4,self.start_neurons*6,(3,3))\n        dense5 = self.denseBlock(conv5,self.start_neurons*6) \n\n        GA =tf.keras.layers.GlobalAveragePooling2D()(dense5)\n        GA= tf.keras.layers.Dropout(0.1)(GA)\n\n        fc = tf.keras.layers.Dense(number_classes,activation=\"softmax\")(GA) #class_types= 10\n\n        self.model = tf.keras.Model(input_layer, fc)\n        self.model.compile(loss=\"sparse_categorical_crossentropy\", optimizer=tf.keras.optimizers.Adam(self.lr),metrics=[\"acc\"])\n        \n        print(\"model input : \",self.model.input)\n        print(\"model output : \",self.model.output)\n        \n        return self.model\n    \n    def run(self,x_train,x_test):\n        checkpoint_filepath = \"./denseNet_classifier.h5\"\n        checkpoint_callback = tf.keras.callbacks.ModelCheckpoint(\n        checkpoint_filepath,\n        monitor=\"val_loss\",\n        save_best_only=True,\n        save_weights_only=True,\n        min = 'min',\n        verbose=1)\n        self.model.load_weights(\"/kaggle/input/basedensenet/denseNet_classifier.h5\")\n        reduce_lr = tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', mode = 'min',\n                              factor=0.1, patience=5, min_lr=0.00001, verbose=1)\n        history = self.model.fit(\n              generate_data(x_train,self.batch_size,True),\n              validation_data=generate_data(x_test,self.batch_size),\n              epochs=self.epochs,\n              steps_per_epoch=len(x_train)//self.batch_size, \n              validation_steps=len(x_test)//self.batch_size,\n            verbose=1,\n            callbacks=[checkpoint_callback,reduce_lr]\n          )\n        \n        return self.model,history\n  \n","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:57:04.652396Z","iopub.execute_input":"2022-09-17T09:57:04.652753Z","iopub.status.idle":"2022-09-17T09:57:04.672547Z","shell.execute_reply.started":"2022-09-17T09:57:04.652721Z","shell.execute_reply":"2022-09-17T09:57:04.671251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nimport pydicom\n\ntraining_image_path = \"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/\"\nsampleImage = \"c3b05294-29be-46e4-8a51-96fd211e4ca5.dcm\"\n\nsample0 = read_Image(training_image_path+sampleImage)\nplt.imshow(sample0)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:58:22.523620Z","iopub.execute_input":"2022-09-17T09:58:22.524710Z","iopub.status.idle":"2022-09-17T09:58:22.904690Z","shell.execute_reply.started":"2022-09-17T09:58:22.524655Z","shell.execute_reply":"2022-09-17T09:58:22.903749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Original Image shape : \",sample0.shape)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:42:58.097898Z","iopub.execute_input":"2022-09-17T09:42:58.098314Z","iopub.status.idle":"2022-09-17T09:42:58.103387Z","shell.execute_reply.started":"2022-09-17T09:42:58.098279Z","shell.execute_reply":"2022-09-17T09:42:58.102560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Minimum Pixel Value : \",sample0.min() ,\"Maximum Pixel Value : \", sample0.max())","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:43:00.334749Z","iopub.execute_input":"2022-09-17T09:43:00.335094Z","iopub.status.idle":"2022-09-17T09:43:00.341018Z","shell.execute_reply.started":"2022-09-17T09:43:00.335067Z","shell.execute_reply":"2022-09-17T09:43:00.340146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\n\ntrainingName=new_train_dataset[\"patientId\"].values\ntrainingLabel=new_train_dataset[\"class\"].values\n\nskf = StratifiedKFold(n_splits=5)\nskf.get_n_splits(trainingName,trainingLabel)\n\nfoldcount=1\nfor train_index, test_index in skf.split(trainingName,trainingLabel):\n\n    X_train, X_test = new_train_dataset.iloc[train_index], new_train_dataset.iloc[test_index]\n\n    X_train = X_train.sample(frac=1).reset_index(drop=True)\n    \n    X_test = X_test.sample(frac=1).reset_index(drop=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:57:10.792092Z","iopub.execute_input":"2022-09-17T09:57:10.792449Z","iopub.status.idle":"2022-09-17T09:57:10.956295Z","shell.execute_reply.started":"2022-09-17T09:57:10.792418Z","shell.execute_reply":"2022-09-17T09:57:10.955345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Randoming Selecting dataset shape : \",X_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:43:05.059393Z","iopub.execute_input":"2022-09-17T09:43:05.059753Z","iopub.status.idle":"2022-09-17T09:43:05.066330Z","shell.execute_reply.started":"2022-09-17T09:43:05.059725Z","shell.execute_reply":"2022-09-17T09:43:05.065160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **We have selected the 6045 images from the dataset. Split the dataset into three parts. The training size is 70% , validation data size is 15% and testing data size is 15%**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_data, valid_temp, y_train, y_valid_temp = train_test_split(X_test[\"patientId\"].values, X_test[\"class\"].values, test_size=0.3, random_state=42,stratify=X_test[\"class\"].values)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:57:15.353144Z","iopub.execute_input":"2022-09-17T09:57:15.353507Z","iopub.status.idle":"2022-09-17T09:57:15.367169Z","shell.execute_reply.started":"2022-09-17T09:57:15.353476Z","shell.execute_reply":"2022-09-17T09:57:15.366113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_data, test_data, y_valid, y_test = train_test_split(valid_temp,y_valid_temp, test_size=0.5, random_state=12,stratify=y_valid_temp)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T10:06:37.956135Z","iopub.execute_input":"2022-09-17T10:06:37.956490Z","iopub.status.idle":"2022-09-17T10:06:37.966276Z","shell.execute_reply.started":"2022-09-17T10:06:37.956460Z","shell.execute_reply":"2022-09-17T10:06:37.965201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"training data shape : \",train_data.shape)\nprint(\"validation data shape : \",valid_data.shape)\nprint(\"testing data shape : \",test_data.shape)\n","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:57:23.127323Z","iopub.execute_input":"2022-09-17T09:57:23.127677Z","iopub.status.idle":"2022-09-17T09:57:23.133343Z","shell.execute_reply.started":"2022-09-17T09:57:23.127646Z","shell.execute_reply":"2022-09-17T09:57:23.132235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test[1:5]","metadata":{"execution":{"iopub.status.busy":"2022-09-17T10:06:54.162012Z","iopub.execute_input":"2022-09-17T10:06:54.162581Z","iopub.status.idle":"2022-09-17T10:06:54.169070Z","shell.execute_reply.started":"2022-09-17T10:06:54.162546Z","shell.execute_reply":"2022-09-17T10:06:54.167885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Change the training and validation dataset for training.**","metadata":{}},{"cell_type":"code","source":"train_new = pd.DataFrame(train_data)\ntrain_new.columns = [\"patientId\"]\ntrain_new[\"class\"] =y_train\n\nvalid_new = pd.DataFrame(valid_data)\nvalid_new.columns = [\"patientId\"]\nvalid_new[\"class\"] =y_valid","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:57:25.868026Z","iopub.execute_input":"2022-09-17T09:57:25.868440Z","iopub.status.idle":"2022-09-17T09:57:25.881619Z","shell.execute_reply.started":"2022-09-17T09:57:25.868406Z","shell.execute_reply":"2022-09-17T09:57:25.880650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_new.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T10:05:08.317979Z","iopub.execute_input":"2022-09-17T10:05:08.318346Z","iopub.status.idle":"2022-09-17T10:05:08.329100Z","shell.execute_reply.started":"2022-09-17T10:05:08.318316Z","shell.execute_reply":"2022-09-17T10:05:08.328072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_new.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T10:05:39.378953Z","iopub.execute_input":"2022-09-17T10:05:39.379627Z","iopub.status.idle":"2022-09-17T10:05:39.391526Z","shell.execute_reply.started":"2022-09-17T10:05:39.379580Z","shell.execute_reply":"2022-09-17T10:05:39.390594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"class names : \",np.unique(train_new[\"class\"].values))","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:57:27.972867Z","iopub.execute_input":"2022-09-17T09:57:27.973232Z","iopub.status.idle":"2022-09-17T09:57:27.980731Z","shell.execute_reply.started":"2022-09-17T09:57:27.973202Z","shell.execute_reply":"2022-09-17T09:57:27.979682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LabelDict ={'Lung Opacity':0,'No Lung Opacity / Not Normal':1,'Normal':2}","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:57:30.698523Z","iopub.execute_input":"2022-09-17T09:57:30.698906Z","iopub.status.idle":"2022-09-17T09:57:30.704460Z","shell.execute_reply.started":"2022-09-17T09:57:30.698871Z","shell.execute_reply":"2022-09-17T09:57:30.703486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LabelDict['No Lung Opacity / Not Normal']","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:57:33.198485Z","iopub.execute_input":"2022-09-17T09:57:33.198859Z","iopub.status.idle":"2022-09-17T09:57:33.205654Z","shell.execute_reply.started":"2022-09-17T09:57:33.198805Z","shell.execute_reply":"2022-09-17T09:57:33.204527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nIMG_WIDTH = IMG_HEIGHT = 256\n\n#batch generator for training\ndef generate_data(train_set, batch_size,shuffle=False):\n\n    i = 0\n    train_ID=train_set[\"patientId\"].values\n    Y_label=train_set[\"class\"].values\n\n    while True:\n        image_batch = np.zeros((batch_size,IMG_WIDTH,IMG_HEIGHT,1))\n        y_batch= np.zeros((batch_size,1))\n\n        for b in range(batch_size):\n            if i == len(train_ID):\n                i = 0\n                #shuffle if u want to\n                if shuffle:\n                    train_set = train_set.sample(frac=1).reset_index(drop=True)\n                    train_ID=train_set[\"patientId\"].values    \n                    Y_label=train_set[\"class\"].values\n\n            sample = training_image_path+train_ID[i]\n            \n            #img=cv2.imread(sample)\n            img = read_Image(sample+\".dcm\")\n            image_resized = cv2.resize(img, (IMG_WIDTH,IMG_HEIGHT))\n            image_batch[b,:,:,0] = image_resized\n            y_batch[b] = LabelDict[Y_label[i]]\n            \n            #image_batch.append(image_resized)            \n            #y_batch.append([LabelDict[Y_label[i]]])\n            i += 1\n        image_batch=np.array(image_batch)\n        y_batch= np.array(y_batch)    \n        yield  image_batch, y_batch","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:57:37.935041Z","iopub.execute_input":"2022-09-17T09:57:37.935399Z","iopub.status.idle":"2022-09-17T09:57:38.108741Z","shell.execute_reply.started":"2022-09-17T09:57:37.935366Z","shell.execute_reply":"2022-09-17T09:57:38.107860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = simpleDensenet(epochs=1,batch_size=32,lr=0.0001)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:57:42.835274Z","iopub.execute_input":"2022-09-17T09:57:42.835656Z","iopub.status.idle":"2022-09-17T09:57:42.840585Z","shell.execute_reply.started":"2022-09-17T09:57:42.835620Z","shell.execute_reply":"2022-09-17T09:57:42.839578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp_model = model.build(inputShape=(256,256,1),number_classes=3)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:57:44.696590Z","iopub.execute_input":"2022-09-17T09:57:44.696964Z","iopub.status.idle":"2022-09-17T09:57:48.834927Z","shell.execute_reply.started":"2022-09-17T09:57:44.696932Z","shell.execute_reply":"2022-09-17T09:57:48.833854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_trained,history = model.run(train_new,valid_new)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T09:58:28.154884Z","iopub.execute_input":"2022-09-17T09:58:28.155870Z","iopub.status.idle":"2022-09-17T10:00:53.032711Z","shell.execute_reply.started":"2022-09-17T09:58:28.155800Z","shell.execute_reply":"2022-09-17T10:00:53.031659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,5))\nplt.subplot(1,3,1)\nplt.plot(range(history.epoch[-1]+1),history.history['loss'],label='Train_Loss')\nplt.plot(range(history.epoch[-1]+1),history.history['val_loss'],label='Val_loss')\nplt.title('LOSS'); plt.xlabel('Epoch'); plt.ylabel('loss');plt.legend();\n\nplt.subplot(1,3,2)\nplt.plot(range(history.epoch[-1]+1),history.history['acc'],label='accuracy')\nplt.plot(range(history.epoch[-1]+1),history.history['val_acc'],label='val_accuracy')\nplt.title('Accuracy'); plt.xlabel('Epoch'); plt.ylabel('accuracy');plt.legend(); ","metadata":{"execution":{"iopub.status.busy":"2022-09-17T10:01:56.710550Z","iopub.status.idle":"2022-09-17T10:01:56.713199Z","shell.execute_reply.started":"2022-09-17T10:01:56.712943Z","shell.execute_reply":"2022-09-17T10:01:56.712970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import itertools\n\ndef plot_confusion_matrix(cm, classes,\n                          normalize=False,\n                          title='Confusion matrix',\n                          cmap=plt.cm.Blues):\n\n    plt.imshow(cm, interpolation='nearest', cmap=cmap)\n    plt.title(title)\n    plt.colorbar()\n    tick_marks = np.arange(len(classes))\n    plt.xticks(tick_marks, classes, rotation=45)\n    plt.yticks(tick_marks, classes)\n\n    if normalize:\n        cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n        print(\"Normalized confusion matrix\")\n    else:\n        print('Confusion matrix, without normalization')\n\n    thresh = cm.max() / 2.\n    for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])):\n        plt.text(j, i, cm[i, j],\n                 horizontalalignment=\"center\",\n                 color=\"white\" if cm[i, j] > thresh else \"black\")\n\n    plt.tight_layout()\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')","metadata":{"execution":{"iopub.status.busy":"2022-09-17T10:01:56.714679Z","iopub.status.idle":"2022-09-17T10:01:56.715460Z","shell.execute_reply.started":"2022-09-17T10:01:56.715191Z","shell.execute_reply":"2022-09-17T10:01:56.715216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n\n#test_ID=X_train[\"patientId\"].values\n#Y_label=X_train[\"class\"].values\n\ny_test_new = []\ny_pred_label = []\nimage_batch = np.zeros((1,IMG_WIDTH,IMG_HEIGHT,1))\n\nfor i in range(len(test_data)):\n    sample = training_image_path+test_data[i]\n            \n    img = read_Image(sample+\".dcm\")\n\n    image_resized = cv2.resize(img, (IMG_WIDTH,IMG_HEIGHT))\n    image_batch[0,:,:,0] = image_resized\n    y_test_new.append(LabelDict[y_test[i]])\n    \n    predicted = model_trained.predict(image_batch)\n    \n    position = np.argmax(predicted)\n    y_pred_label.append(position)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T10:09:09.860996Z","iopub.execute_input":"2022-09-17T10:09:09.861469Z","iopub.status.idle":"2022-09-17T10:09:59.591208Z","shell.execute_reply.started":"2022-09-17T10:09:09.861428Z","shell.execute_reply":"2022-09-17T10:09:59.590245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Calculate the metrics of deep learning model**","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import classification_report\n#classification report\nprint(\"Deep Learning Based\")\nprint(classification_report(y_test_new, y_pred_label))","metadata":{"execution":{"iopub.status.busy":"2022-09-17T10:10:22.553571Z","iopub.execute_input":"2022-09-17T10:10:22.553963Z","iopub.status.idle":"2022-09-17T10:10:22.568019Z","shell.execute_reply.started":"2022-09-17T10:10:22.553931Z","shell.execute_reply":"2022-09-17T10:10:22.567074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nclass_list = [str(i) for i in range(3)]\ncm = confusion_matrix(y_test_new, y_pred_label)\nplot_confusion_matrix(cm, classes=class_list)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T10:10:39.043460Z","iopub.execute_input":"2022-09-17T10:10:39.044205Z","iopub.status.idle":"2022-09-17T10:10:39.306305Z","shell.execute_reply.started":"2022-09-17T10:10:39.044166Z","shell.execute_reply":"2022-09-17T10:10:39.304864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, recall_score, precision_score, auc, roc_auc_score, roc_curve,f1_score\nfrom sklearn.metrics import confusion_matrix, classification_report, f1_score\n\ncomparison_dict = {}\ndef metric_calculation(trainLabel,y_pred,classifier_name_abv):    \n    \n    comparison_dict[classifier_name_abv] = [\n        accuracy_score(trainLabel, y_pred),\n        precision_score(trainLabel, y_pred,average=\"macro\"),\n        recall_score(trainLabel, y_pred,average=\"macro\"),\n        f1_score(trainLabel, y_pred,average=\"macro\")\n\n    ]\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-09-17T10:22:02.898023Z","iopub.execute_input":"2022-09-17T10:22:02.898414Z","iopub.status.idle":"2022-09-17T10:22:02.904913Z","shell.execute_reply.started":"2022-09-17T10:22:02.898381Z","shell.execute_reply":"2022-09-17T10:22:02.903895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test_new = np.array(y_test_new)\ny_test_new","metadata":{"execution":{"iopub.status.busy":"2022-09-17T10:21:04.495749Z","iopub.execute_input":"2022-09-17T10:21:04.496259Z","iopub.status.idle":"2022-09-17T10:21:04.510957Z","shell.execute_reply.started":"2022-09-17T10:21:04.496222Z","shell.execute_reply":"2022-09-17T10:21:04.509719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_label=np.array(y_pred_label)\ny_pred_label","metadata":{"execution":{"iopub.status.busy":"2022-09-17T10:21:33.578571Z","iopub.execute_input":"2022-09-17T10:21:33.578969Z","iopub.status.idle":"2022-09-17T10:21:33.588084Z","shell.execute_reply.started":"2022-09-17T10:21:33.578934Z","shell.execute_reply":"2022-09-17T10:21:33.586916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metric_calculation(y_test_new,y_pred_label,classifier_name_abv=\"DenseNet\")","metadata":{"execution":{"iopub.status.busy":"2022-09-17T10:22:33.442506Z","iopub.execute_input":"2022-09-17T10:22:33.442891Z","iopub.status.idle":"2022-09-17T10:22:33.453000Z","shell.execute_reply.started":"2022-09-17T10:22:33.442857Z","shell.execute_reply":"2022-09-17T10:22:33.452010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Machine Learning classifier based classify the image**","metadata":{}},{"cell_type":"code","source":"def converimage2Feature(data_input):\n\n    test_ID=data_input[\"patientId\"].values\n    Y_label=data_input[\"class\"].values\n\n    train_image_feature = []\n    Y_label_new  =[]\n    image_batch = np.zeros((1,IMG_WIDTH,IMG_HEIGHT,1))\n\n    for i in range(len(test_ID)):\n        sample = training_image_path+test_ID[i]\n            \n        img = read_Image(sample+\".dcm\")\n        image_resized = cv2.resize(img, (32,32))\n        train_image_feature.append(image_resized.flatten())\n        Y_label_new.append(LabelDict[Y_label[i]])\n    return np.array(train_image_feature),np.array(Y_label_new)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T10:28:38.512486Z","iopub.execute_input":"2022-09-17T10:28:38.512869Z","iopub.status.idle":"2022-09-17T10:28:38.519674Z","shell.execute_reply.started":"2022-09-17T10:28:38.512813Z","shell.execute_reply":"2022-09-17T10:28:38.518668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_feature,y_train_feat = converimage2Feature(train_new)\n\nvalid_feature,y_valid_feat = converimage2Feature(valid_new)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T10:28:41.685383Z","iopub.execute_input":"2022-09-17T10:28:41.686454Z","iopub.status.idle":"2022-09-17T10:29:42.704089Z","shell.execute_reply.started":"2022-09-17T10:28:41.686412Z","shell.execute_reply":"2022-09-17T10:29:42.703111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from xgboost import XGBClassifier\n\n\"\"\"\nspace={'max_depth': hp.quniform(\"max_depth\", 3, 18, 1),\n        'gamma': hp.uniform ('gamma', 1,9),\n        'reg_alpha' : hp.quniform('reg_alpha', 40,180,1),\n        'reg_lambda' : hp.uniform('reg_lambda', 0,1),\n        'colsample_bytree' : hp.uniform('colsample_bytree', 0.5,1),\n        'min_child_weight' : hp.quniform('min_child_weight', 0, 10, 1),\n        'n_estimators': 180,\n        'seed': 0\n    }\n\"\"\"\nclf=XGBClassifier(\n                n_estimators =1000)\n\nevaluation = [( train_feature, y_train_feat), ( valid_feature, y_valid_feat)]\n\nclf.fit(train_feature, y_train_feat,\n        eval_set=evaluation, eval_metric=\"auc\",\n        early_stopping_rounds=10,verbose=False)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T10:32:27.175445Z","iopub.execute_input":"2022-09-17T10:32:27.175894Z","iopub.status.idle":"2022-09-17T10:35:10.542178Z","shell.execute_reply.started":"2022-09-17T10:32:27.175854Z","shell.execute_reply":"2022-09-17T10:35:10.541256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_image_feature = []\n#Y_label_new  =[]\nfor i in range(len(test_data)):\n    sample = training_image_path+test_data[i]\n\n    img = read_Image(sample+\".dcm\")\n    image_resized = cv2.resize(img, (32,32))\n    test_image_feature.append(image_resized.flatten())\n    #Y_label_new.append(LabelDict[Y_label[i]])","metadata":{"execution":{"iopub.status.busy":"2022-09-17T11:01:22.225058Z","iopub.execute_input":"2022-09-17T11:01:22.225979Z","iopub.status.idle":"2022-09-17T11:01:34.300437Z","shell.execute_reply.started":"2022-09-17T11:01:22.225941Z","shell.execute_reply":"2022-09-17T11:01:34.299287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xg_test_pred = clf.predict(test_image_feature)\nxg_test_pred","metadata":{"execution":{"iopub.status.busy":"2022-09-17T11:02:38.493400Z","iopub.execute_input":"2022-09-17T11:02:38.493755Z","iopub.status.idle":"2022-09-17T11:02:38.535083Z","shell.execute_reply.started":"2022-09-17T11:02:38.493723Z","shell.execute_reply":"2022-09-17T11:02:38.534232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Calculate the metrics of XGBoost model**","metadata":{}},{"cell_type":"code","source":"metric_calculation(y_test_new,xg_test_pred,\"XGBoost\")","metadata":{"execution":{"iopub.status.busy":"2022-09-17T11:04:34.275572Z","iopub.execute_input":"2022-09-17T11:04:34.276139Z","iopub.status.idle":"2022-09-17T11:04:34.287669Z","shell.execute_reply.started":"2022-09-17T11:04:34.276097Z","shell.execute_reply":"2022-09-17T11:04:34.286745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_list = [str(i) for i in range(3)]\ncm = confusion_matrix(y_test_new, xg_test_pred)\nplot_confusion_matrix(cm, classes=class_list)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T11:04:35.984727Z","iopub.execute_input":"2022-09-17T11:04:35.985353Z","iopub.status.idle":"2022-09-17T11:04:36.282713Z","shell.execute_reply.started":"2022-09-17T11:04:35.985305Z","shell.execute_reply":"2022-09-17T11:04:36.281055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Creating the training metrics datasheet\ncomparison_matrix = {}\nfor key, value in comparison_dict.items():\n    comparison_matrix[str(key)] = value[0:5]\n\ntestDataMetrics = pd.DataFrame(comparison_matrix,\n                            index=['Accuracy', 'Precision', 'Recall','F1_score']).T\n\ntestDataMetrics.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T11:04:40.000351Z","iopub.execute_input":"2022-09-17T11:04:40.000706Z","iopub.status.idle":"2022-09-17T11:04:40.014400Z","shell.execute_reply.started":"2022-09-17T11:04:40.000674Z","shell.execute_reply":"2022-09-17T11:04:40.013342Z"},"trusted":true},"execution_count":null,"outputs":[]}]}