{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import pandas as pd \nimport numpy as np\nimport matplotlib\nimport matplotlib.pyplot as plt\nfrom matplotlib.patches import Rectangle\nimport seaborn as sns\nimport pydicom as dcm\n%matplotlib inline\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_info_df = pd.read_csv('../input/rsna-pneumonia-detection-challenge/stage_2_detailed_class_info.csv')\ntrain_labels_df = pd.read_csv('../input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv') ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f\"Detailed class info -  rows: {class_info_df.shape[0]}, columns: {class_info_df.shape[1]}\")\nprint(f\"Train labels -  rows: {train_labels_df.shape[0]}, columns: {train_labels_df.shape[1]}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_info_df.head(10)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Class Info provides details on Patients' lung opacity which is used to analyse if the patient is tested +ve or -ve to Pnuemonia"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_labels_df.head(10)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Train labels gives the Image coordinates of the Chest which form part of the evidences for Pnuemonia"},{"metadata":{},"cell_type":"markdown","source":"EDA in Dataset\n\n\nChecking the missing data from dataset"},{"metadata":{"trusted":true},"cell_type":"code","source":"def missing_data(data):\n    total = data.isnull().sum().sort_values(ascending = False)\n    percent = (data.isnull().sum()/data.isnull().count()*100).sort_values(ascending = False)\n    return np.transpose(pd.concat([total, percent], axis=1, keys=['Total', 'Percent']))\nmissing_data(train_labels_df)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"missing_data(class_info_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f, ax = plt.subplots(1,1, figsize=(6,4))\ntotal = float(len(class_info_df))\nsns.countplot(class_info_df['class'],order = class_info_df['class'].value_counts().index, palette='Set3')\nfor p in ax.patches:\n    height = p.get_height()\n    ax.text(p.get_x()+p.get_width()/2.,\n            height + 3,\n            '{:1.2f}%'.format(100*height/total),\n            ha=\"center\") \nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Getting the proportion of each class against the total data"},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_feature_distribution(data, feature):\n    # Get the count for each label\n    label_counts = data[feature].value_counts()\n\n    # Get total number of samples\n    total_samples = len(data)\n\n    # Count the number of items in each class\n    print(\"Feature: {}\".format(feature))\n    for i in range(len(label_counts)):\n        label = label_counts.index[i]\n        count = label_counts.values[i]\n        percent = int((count / total_samples) * 10000) / 100\n        print(\"{:<30s}:   {} or {}%\".format(label, count, percent))\n\nget_feature_distribution(class_info_df, 'class')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_class_df = train_labels_df.merge(class_info_df, left_on='patientId', right_on='patientId', how='inner')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_class_df.head(5)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Grouping the data based on the Target class"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(nrows=1,figsize=(12,6))\ntmp = train_class_df.groupby('Target')['class'].value_counts()\ndf = pd.DataFrame(data={'Exams': tmp.values}, index=tmp.index).reset_index()\nsns.barplot(ax=ax,x = 'Target', y='Exams',hue='class',data=df, palette='Set3')\nplt.title(\"Chest exams class and Target\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Traget class 1 denotes Lung Opacity or detected with Pnuemonia\n\nTarget class 0 denotes no Lung Opacity/not normal/Normal or not detected with Pneumonia"},{"metadata":{},"cell_type":"markdown","source":"Lung opacity  - Target class 1 : Plotting Density plot for the image coordinates "},{"metadata":{"trusted":true},"cell_type":"code","source":"target1 = train_class_df[train_class_df['Target']==1]\nsns.set_style('whitegrid')\nplt.figure()\nfig, ax = plt.subplots(2,2,figsize=(12,12))\nsns.distplot(target1['x'],kde=True,bins=50, color=\"red\", ax=ax[0,0])\nsns.distplot(target1['y'],kde=True,bins=50, color=\"blue\", ax=ax[0,1])\nsns.distplot(target1['width'],kde=True,bins=50, color=\"green\", ax=ax[1,0])\nsns.distplot(target1['height'],kde=True,bins=50, color=\"magenta\", ax=ax[1,1])\nlocs, labels = plt.xticks()\nplt.tick_params(axis='both', which='major', labelsize=12)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Sample of Centrepoints shown using the rectangle from the average of coordinates X and width as XC, Y and height as YC"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(1,1,figsize=(7,7))\ntarget_sample = target1.sample(2000)\ntarget_sample['xc'] = target_sample['x'] + target_sample['width'] / 2\ntarget_sample['yc'] = target_sample['y'] + target_sample['height'] / 2\nplt.title(\"Centers of Lung Opacity rectangles (brown) over rectangles (blue)\\nSample size: 2000\")\ntarget_sample.plot.scatter(x='xc', y='yc', xlim=(0,1024), ylim=(0,1024), ax=ax, alpha=0.8, marker=\".\", color=\"brown\")\nfor i, crt_sample in target_sample.iterrows():\n    ax.add_patch(Rectangle(xy=(crt_sample['x'], crt_sample['y']),\n                width=crt_sample['width'],height=crt_sample['height'],alpha=3.5e-3, color=\"blue\"))\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"DICOM Data exploration"},{"metadata":{"trusted":true},"cell_type":"code","source":"image_sample_path = os.listdir('../input/rsna-pneumonia-detection-challenge/stage_2_train_images')[:5]\nprint(image_sample_path)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Each File name is identified by the Patient IDs\n\nChecking the count of images in Train and Test folders"},{"metadata":{"trusted":true},"cell_type":"code","source":"image_train_path = os.listdir('../input/rsna-pneumonia-detection-challenge/stage_2_train_images')\nimage_test_path = os.listdir('../input/rsna-pneumonia-detection-challenge/stage_2_test_images')\nprint(\"Number of images in train set:\", len(image_train_path),\"\\nNumber of images in test set:\", len(image_test_path))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"There may be duplicates existing in the dataset which would have resulted in reduction in the numbers of images in Training set from \n    30227 to 26684. We can check the same below for existance of duplicates"},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Unique patientId in  train_class_df: \", train_class_df['patientId'].nunique())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Checking the distribution of duplicate entries in Target classes"},{"metadata":{},"cell_type":"markdown","source":"DICOM Indormation from Image"},{"metadata":{"trusted":true},"cell_type":"code","source":"samplePatientID = list(train_class_df[:3].T.to_dict().values())[0]['patientId']\nsamplePatientID = samplePatientID+'.dcm'\ndicom_file_path = os.path.join('../input/rsna-pneumonia-detection-challenge/stage_2_train_images',samplePatientID)\ndicom_file_dataset = dcm.read_file(dicom_file_path)\ndicom_file_dataset","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"From the above we can understand the details like:\n\nPatient ID\n\nBirth date\n\nSex \n\nAge\n\nExamined body part\n\nno of rows and columns etc...."},{"metadata":{},"cell_type":"markdown","source":"Sampling of Images with Target class 1"},{"metadata":{"trusted":true},"cell_type":"code","source":"def show_dicom_images(data):\n    img_data = list(data.T.to_dict().values())\n    f, ax = plt.subplots(3,3, figsize=(16,18))\n    for i,data_row in enumerate(img_data):\n        patientImage = data_row['patientId']+'.dcm'\n        imagePath = os.path.join('../input/rsna-pneumonia-detection-challenge/stage_2_train_images',patientImage)\n        data_row_img_data = dcm.read_file(imagePath)\n        modality = data_row_img_data.Modality\n        age = data_row_img_data.PatientAge\n        sex = data_row_img_data.PatientSex\n        data_row_img = dcm.dcmread(imagePath)\n        ax[i//3, i%3].imshow(data_row_img.pixel_array, cmap=plt.cm.bone) \n        ax[i//3, i%3].axis('off')\n        ax[i//3, i%3].set_title('ID: {}\\nModality: {} Age: {} Sex: {} Target: {}\\nClass: {}\\nWindow: {}:{}:{}:{}'.format(\n                data_row['patientId'],\n                modality, age, sex, data_row['Target'], data_row['class'], \n                data_row['x'],data_row['y'],data_row['width'],data_row['height']))\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"show_dicom_images(train_class_df[train_class_df['Target']==1].sample(9))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Parsing the whole dataset with Target 1 and gather all coordinates showing the lung opacity in the image "},{"metadata":{"trusted":true},"cell_type":"code","source":"def show_dicom_images_with_boxes(data):\n    img_data = list(data.T.to_dict().values())\n    f, ax = plt.subplots(3,3, figsize=(16,18))\n    for i,data_row in enumerate(img_data):\n        patientImage = data_row['patientId']+'.dcm'\n        imagePath = os.path.join('../input/rsna-pneumonia-detection-challenge/stage_2_train_images',patientImage)\n        data_row_img_data = dcm.read_file(imagePath)\n        modality = data_row_img_data.Modality\n        age = data_row_img_data.PatientAge\n        sex = data_row_img_data.PatientSex\n        data_row_img = dcm.dcmread(imagePath)\n        ax[i//3, i%3].imshow(data_row_img.pixel_array, cmap=plt.cm.bone) \n        ax[i//3, i%3].axis('off')\n        ax[i//3, i%3].set_title('ID: {}\\nModality: {} Age: {} Sex: {} Target: {}\\nClass: {}'.format(\n                data_row['patientId'],modality, age, sex, data_row['Target'], data_row['class']))\n        rows = train_class_df[train_class_df['patientId']==data_row['patientId']]\n        box_data = list(rows.T.to_dict().values())\n        for j, row in enumerate(box_data):\n            ax[i//3, i%3].add_patch(Rectangle(xy=(row['x'], row['y']),\n                        width=row['width'],height=row['height'], \n                        color=\"yellow\",alpha = 0.1))   \n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"show_dicom_images_with_boxes(train_class_df[train_class_df['Target']==1].sample(9))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"show_dicom_images_with_boxes(train_class_df[train_class_df['Target']==1].sample(9))"},{"metadata":{},"cell_type":"markdown","source":"Images above highlight mutilple lung opacities \n\nChecking sample images with Target = 0"},{"metadata":{"trusted":true},"cell_type":"code","source":"show_dicom_images(train_class_df[train_class_df['Target']==0].sample(9))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}