{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":10338,"databundleVersionId":862042,"sourceType":"competition"}],"dockerImageVersionId":30579,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-14T17:20:54.864799Z","iopub.execute_input":"2023-11-14T17:20:54.865232Z","iopub.status.idle":"2023-11-14T17:20:54.871073Z","shell.execute_reply.started":"2023-11-14T17:20:54.865198Z","shell.execute_reply":"2023-11-14T17:20:54.869694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom glob import glob\nimport math\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport pydicom as dcm\nfrom sklearn.model_selection import train_test_split\n\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Layer, Convolution2D, Flatten, Dense, Concatenate, UpSampling2D, Conv2D, Reshape, GlobalAveragePooling2D\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.applications.mobilenet import MobileNet, preprocess_input\nfrom tensorflow.keras.utils import Sequence\nfrom tensorflow.keras.applications.resnet import ResNet50, preprocess_input as resnetProcess_input\n","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:29:26.304546Z","iopub.execute_input":"2023-11-14T17:29:26.305257Z","iopub.status.idle":"2023-11-14T17:29:26.312837Z","shell.execute_reply.started":"2023-11-14T17:29:26.305208Z","shell.execute_reply":"2023-11-14T17:29:26.311830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reading the labels dataset and displaying the first few records\ncustom_labels = pd.read_csv(\"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv\")\ncustom_labels.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:29:29.254690Z","iopub.execute_input":"2023-11-14T17:29:29.255693Z","iopub.status.idle":"2023-11-14T17:29:29.344889Z","shell.execute_reply.started":"2023-11-14T17:29:29.255645Z","shell.execute_reply":"2023-11-14T17:29:29.343539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"custom_labels.describe()","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:29:58.380060Z","iopub.execute_input":"2023-11-14T17:29:58.380547Z","iopub.status.idle":"2023-11-14T17:29:58.420062Z","shell.execute_reply.started":"2023-11-14T17:29:58.380508Z","shell.execute_reply":"2023-11-14T17:29:58.418939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"custom_labels.info()","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:30:04.515463Z","iopub.execute_input":"2023-11-14T17:30:04.515951Z","iopub.status.idle":"2023-11-14T17:30:04.544971Z","shell.execute_reply.started":"2023-11-14T17:30:04.515906Z","shell.execute_reply":"2023-11-14T17:30:04.544031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"custom_labels.hist()","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:30:19.817271Z","iopub.execute_input":"2023-11-14T17:30:19.817748Z","iopub.status.idle":"2023-11-14T17:30:20.819732Z","shell.execute_reply.started":"2023-11-14T17:30:19.817706Z","shell.execute_reply":"2023-11-14T17:30:20.818422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = custom_labels\nprint(f'Dataframe shape : {custom_labels.shape}')\n","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:31:30.775850Z","iopub.execute_input":"2023-11-14T17:31:30.776357Z","iopub.status.idle":"2023-11-14T17:31:30.782885Z","shell.execute_reply.started":"2023-11-14T17:31:30.776323Z","shell.execute_reply":"2023-11-14T17:31:30.781631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Number of unique Patient Ids :', custom_labels['patientId'].unique().shape[0])\n","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:31:44.400288Z","iopub.execute_input":"2023-11-14T17:31:44.400763Z","iopub.status.idle":"2023-11-14T17:31:44.412950Z","shell.execute_reply.started":"2023-11-14T17:31:44.400728Z","shell.execute_reply":"2023-11-14T17:31:44.411678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df['patientId'] == df[df['patientId'].duplicated()]['patientId'].values[0]]\n","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:31:53.867903Z","iopub.execute_input":"2023-11-14T17:31:53.868438Z","iopub.status.idle":"2023-11-14T17:31:53.904313Z","shell.execute_reply.started":"2023-11-14T17:31:53.868398Z","shell.execute_reply":"2023-11-14T17:31:53.903183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## we can see that all the null column values are with Target 0 indicating that those patients do not have penumonia\ndf[df.isnull().any(axis=1)].Target.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:32:26.342423Z","iopub.execute_input":"2023-11-14T17:32:26.342877Z","iopub.status.idle":"2023-11-14T17:32:26.358948Z","shell.execute_reply.started":"2023-11-14T17:32:26.342846Z","shell.execute_reply":"2023-11-14T17:32:26.357637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## we can see that all the non null column values are with Target 1 indicating that those patients have pneumonia\ndf[~df.isnull().any(axis=1)].Target.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:32:39.685541Z","iopub.execute_input":"2023-11-14T17:32:39.686015Z","iopub.status.idle":"2023-11-14T17:32:39.701107Z","shell.execute_reply.started":"2023-11-14T17:32:39.685979Z","shell.execute_reply":"2023-11-14T17:32:39.699750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Distubution of Targets , there are 20672 records with no pneumonia and 9555 with pneumonia\ndf.Target.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:32:56.669327Z","iopub.execute_input":"2023-11-14T17:32:56.669864Z","iopub.status.idle":"2023-11-14T17:32:56.680737Z","shell.execute_reply.started":"2023-11-14T17:32:56.669811Z","shell.execute_reply":"2023-11-14T17:32:56.679363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Assuming df is your DataFrame containing the 'Target' column\ntarget_counts = df['Target'].value_counts()\n\n# Plotting the distribution\nplt.figure(figsize=(8, 6))\ntarget_counts.plot(kind='bar', color=['skyblue', 'orange'])\nplt.title('Distribution of Targets')\nplt.xlabel('Target')\nplt.ylabel('Count')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:33:35.781814Z","iopub.execute_input":"2023-11-14T17:33:35.782305Z","iopub.status.idle":"2023-11-14T17:33:36.048349Z","shell.execute_reply.started":"2023-11-14T17:33:35.782264Z","shell.execute_reply":"2023-11-14T17:33:36.046962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Assuming df is your DataFrame containing the 'Target' column\ntarget_counts = df['Target'].value_counts()\n\n# Plotting the distribution as a pie chart\nplt.figure(figsize=(8, 8))\nplt.pie(target_counts, labels=target_counts.index, autopct='%1.1f%%', colors=['skyblue', 'orange'])\nplt.title('Distribution of Targets')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:34:07.931057Z","iopub.execute_input":"2023-11-14T17:34:07.931504Z","iopub.status.idle":"2023-11-14T17:34:08.105146Z","shell.execute_reply.started":"2023-11-14T17:34:07.931471Z","shell.execute_reply":"2023-11-14T17:34:08.103952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['x'] = df['x'].fillna(0)\ndf['y'] = df['y'].fillna(0)\ndf['width'] = df['width'].fillna(0)\ndf['height'] = df['height'].fillna(0)","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:36:31.233290Z","iopub.execute_input":"2023-11-14T17:36:31.233804Z","iopub.status.idle":"2023-11-14T17:36:31.245425Z","shell.execute_reply.started":"2023-11-14T17:36:31.233765Z","shell.execute_reply":"2023-11-14T17:36:31.244100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reading the class Info Data Set","metadata":{}},{"cell_type":"code","source":"## Reading the classes label , \nclass_labels = pd.read_csv('/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_detailed_class_info.csv')\nclass_labels.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:38:31.995178Z","iopub.execute_input":"2023-11-14T17:38:31.995699Z","iopub.status.idle":"2023-11-14T17:38:32.070086Z","shell.execute_reply.started":"2023-11-14T17:38:31.995664Z","shell.execute_reply":"2023-11-14T17:38:32.069122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_labels.shape\n","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:38:38.477204Z","iopub.execute_input":"2023-11-14T17:38:38.477707Z","iopub.status.idle":"2023-11-14T17:38:38.486013Z","shell.execute_reply.started":"2023-11-14T17:38:38.477669Z","shell.execute_reply":"2023-11-14T17:38:38.484851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_labels.info()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:38:43.704116Z","iopub.execute_input":"2023-11-14T17:38:43.705210Z","iopub.status.idle":"2023-11-14T17:38:43.722036Z","shell.execute_reply.started":"2023-11-14T17:38:43.705173Z","shell.execute_reply":"2023-11-14T17:38:43.720838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_labels['class'].value_counts()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:38:54.694509Z","iopub.execute_input":"2023-11-14T17:38:54.694956Z","iopub.status.idle":"2023-11-14T17:38:54.706336Z","shell.execute_reply.started":"2023-11-14T17:38:54.694926Z","shell.execute_reply":"2023-11-14T17:38:54.705009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Assuming df is your DataFrame containing the 'Target' column\ntarget_counts = class_labels['class'].value_counts()\n\n# Plotting the distribution\nplt.figure(figsize=(8, 6))\ntarget_counts.plot(kind='bar', color=['skyblue', 'orange'])\nplt.title('Distribution of Targets')\nplt.xlabel('Target')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:39:22.864125Z","iopub.execute_input":"2023-11-14T17:39:22.864599Z","iopub.status.idle":"2023-11-14T17:39:23.134893Z","shell.execute_reply.started":"2023-11-14T17:39:22.864551Z","shell.execute_reply":"2023-11-14T17:39:23.133619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Assuming df is your DataFrame containing the 'Target' column\ntarget_counts = class_labels['class'].value_counts()\n\n# Plotting the distribution as a pie chart\nplt.figure(figsize=(8, 8))\nplt.pie(target_counts, labels=target_counts.index, autopct='%1.1f%%', colors=['skyblue', 'orange','green'])\nplt.title('Distribution of Targets')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:39:55.702817Z","iopub.execute_input":"2023-11-14T17:39:55.703272Z","iopub.status.idle":"2023-11-14T17:39:55.886848Z","shell.execute_reply.started":"2023-11-14T17:39:55.703238Z","shell.execute_reply":"2023-11-14T17:39:55.885003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Let's concatenate the two dataframes and use the merged dataframe for further analysis.\n\n","metadata":{}},{"cell_type":"code","source":"final_df = pd.concat([df, class_labels['class']], axis = 1)\n\nfinal_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:40:48.173604Z","iopub.execute_input":"2023-11-14T17:40:48.174105Z","iopub.status.idle":"2023-11-14T17:40:48.195167Z","shell.execute_reply.started":"2023-11-14T17:40:48.174065Z","shell.execute_reply":"2023-11-14T17:40:48.194259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating subplots and setting figure size\ncustom_fig, custom_ax = plt.subplots(nrows=1, figsize=(12, 6))\n\n# Grouping and aggregating data\ncustom_temp = final_df.groupby('Target')['class'].value_counts()\ncustom_data_target_class = pd.DataFrame(data={'Values': custom_temp.values}, index=custom_temp.index).reset_index()\n\n# Creating a bar plot with seaborn\nsns.barplot(ax=custom_ax, x='Target', y='Values', hue='class', data=custom_data_target_class, palette='Set3')\n\n# Adding title to the plot\nplt.title('Custom Class and Target Distribution')\n","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:41:53.255981Z","iopub.execute_input":"2023-11-14T17:41:53.256451Z","iopub.status.idle":"2023-11-14T17:41:53.665863Z","shell.execute_reply.started":"2023-11-14T17:41:53.256413Z","shell.execute_reply":"2023-11-14T17:41:53.664653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport pydicom as dcm\nimport math\n\ndef visualize_images(data,type_of):\n    img_data = list(data.T.to_dict().values())\n    fig, ax = plt.subplots(2, 2, figsize=(12, 12))\n\n    for i, data_row in enumerate(img_data[:4]):  # Plot only the first 4 images\n        patient_id = data_row['patientId']\n        dcm_file = '/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/{}.dcm'.format(patient_id)\n        data_row_img_data = dcm.read_file(dcm_file)\n        modality = data_row_img_data.Modality\n        age = data_row_img_data.PatientAge\n        sex = data_row_img_data.PatientSex\n        data_row_img = dcm.dcmread(dcm_file)\n\n        ax[i // 2, i % 2].imshow(data_row_img.pixel_array, cmap=plt.cm.bone)\n        ax[i // 2, i % 2].axis('off')\n        ax[i // 2, i % 2].set_title('{} \\nInfo: Age: {} Sex: {} \\n {}'.format(type_of,age, sex, data_row['class']))\n\n        # Draw bounding box if available\n        if not math.isnan(data_row['x']):\n            x, y, width, height = data_row['x'], data_row['y'], data_row['width'], data_row['height']\n            rect = patches.Rectangle((x, y), width, height, linewidth=2, edgecolor='r', facecolor='none')\n            ax[i // 2, i % 2].add_patch(rect)\n\n    plt.tight_layout()\n    plt.show()\n\n# Example usage:\n# visualize_images(your_data_frame)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:52:26.743069Z","iopub.execute_input":"2023-11-14T17:52:26.743722Z","iopub.status.idle":"2023-11-14T17:52:26.755517Z","shell.execute_reply.started":"2023-11-14T17:52:26.743677Z","shell.execute_reply":"2023-11-14T17:52:26.754071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Displaying Chest X-ray Images of Patients who have Pneuomina","metadata":{}},{"cell_type":"code","source":"## checking few images which has pneuonia \nvisualize_images(final_df[final_df['Target']==1].sample(4),\"Pneuomina\")","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:52:27.242546Z","iopub.execute_input":"2023-11-14T17:52:27.243020Z","iopub.status.idle":"2023-11-14T17:52:28.748633Z","shell.execute_reply.started":"2023-11-14T17:52:27.242986Z","shell.execute_reply":"2023-11-14T17:52:28.747304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Displaying Chest X-ray Images of Patients who not have Pneuomina","metadata":{}},{"cell_type":"code","source":"## checking few images which has pneuonia \nvisualize_images(final_df[final_df['Target']==0].sample(4),\"No Pneuomina\")","metadata":{"execution":{"iopub.status.busy":"2023-11-14T17:53:10.883293Z","iopub.execute_input":"2023-11-14T17:53:10.883762Z","iopub.status.idle":"2023-11-14T17:53:12.369212Z","shell.execute_reply.started":"2023-11-14T17:53:10.883726Z","shell.execute_reply":"2023-11-14T17:53:12.367989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}