{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport pydicom\nimport cv2\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\nimport seaborn as sns\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import classification_report\nfrom sklearn.metrics import roc_curve\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import precision_recall_curve\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import OneHotEncoder\nfrom tqdm import tqdm_notebook\nfrom matplotlib.patches import Rectangle\n\nfrom sklearn.model_selection import GridSearchCV;\nfrom sklearn.impute import SimpleImputer;\nfrom sklearn.model_selection import train_test_split, GridSearchCV, StratifiedKFold\n#from imblearn.pipeline import Pipeline as imbpipeline\nfrom sklearn.pipeline import Pipeline\n\n# machine learning\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC, LinearSVC\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.linear_model import Perceptron\nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nimport math\nimport random\n\nimport h5py\n\nimport tensorflow\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras import regularizers, optimizers\n\nimport os\nimport zipfile\nimport cv2\n\nfrom tensorflow.keras import backend as kbackend\n#from tensorflow.keras.backend import set_session\nfrom tensorflow.keras.applications.mobilenet import MobileNet\nfrom tensorflow.keras.layers import Dense, Dropout # construct each layer\nfrom tensorflow.keras.layers import MaxPooling2D # swipe across by pool size\nfrom tensorflow.keras.layers import Flatten, GlobalAveragePooling2D\nfrom tensorflow.keras.layers import Concatenate, Conv2D, Reshape, UpSampling2D\nfrom tensorflow.keras.losses import binary_crossentropy\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.optimizers import Adam\n\nfrom sklearn.metrics import classification_report\n\nfrom sklearn.metrics import precision_recall_fscore_support as prfscore\n\nfrom tensorflow.keras.applications.mobilenet import preprocess_input\n\nimport pydicom","metadata":{"tags":[],"execution":{"iopub.status.busy":"2022-10-09T13:08:36.417666Z","iopub.execute_input":"2022-10-09T13:08:36.418341Z","iopub.status.idle":"2022-10-09T13:08:37.099727Z","shell.execute_reply.started":"2022-10-09T13:08:36.418089Z","shell.execute_reply":"2022-10-09T13:08:37.098844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 1: Import the data. ","metadata":{}},{"cell_type":"code","source":"classInfodf = pd.read_csv('/kaggle/input/stage_2_detailed_class_info.csv')","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:37.104125Z","iopub.execute_input":"2022-10-09T13:08:37.106372Z","iopub.status.idle":"2022-10-09T13:08:37.163206Z","shell.execute_reply.started":"2022-10-09T13:08:37.10631Z","shell.execute_reply":"2022-10-09T13:08:37.162452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classInfodf = pd.read_csv('/kaggle/input/stage_2_detailed_class_info.csv')\ntrainlabelsdf = pd.read_csv(\"/kaggle/input/stage_2_train_labels.csv\")\ntrainImagesPath = Path(\"/kaggle/input/stage_2_train_images\")\ntestImagesPath = Path(\"/kaggle/input/stage_2_test_images\")\n\nsampleSubPath = Path(\"/kaggle/input/stage_2_sample_submission.csv\")\nimport os;\nimage_train_path = os.listdir(trainImagesPath)\nimage_test_path = os.listdir(testImagesPath)\n\n                                                                    ","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:37.167412Z","iopub.execute_input":"2022-10-09T13:08:37.169519Z","iopub.status.idle":"2022-10-09T13:08:37.281757Z","shell.execute_reply.started":"2022-10-09T13:08:37.169451Z","shell.execute_reply":"2022-10-09T13:08:37.280941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### <font size=4 color='blue' > EDA </font>","metadata":{}},{"cell_type":"code","source":"classInfodf.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:37.286061Z","iopub.execute_input":"2022-10-09T13:08:37.288159Z","iopub.status.idle":"2022-10-09T13:08:37.316338Z","shell.execute_reply.started":"2022-10-09T13:08:37.288106Z","shell.execute_reply":"2022-10-09T13:08:37.315669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classInfodf.info()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:37.31994Z","iopub.execute_input":"2022-10-09T13:08:37.321978Z","iopub.status.idle":"2022-10-09T13:08:37.34041Z","shell.execute_reply.started":"2022-10-09T13:08:37.321927Z","shell.execute_reply":"2022-10-09T13:08:37.339654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4 color='blue' > There are two features 1. Patient ID 2. Class</font>","metadata":{}},{"cell_type":"code","source":"classInfodf.patientId.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:37.344306Z","iopub.execute_input":"2022-10-09T13:08:37.346607Z","iopub.status.idle":"2022-10-09T13:08:37.366898Z","shell.execute_reply.started":"2022-10-09T13:08:37.346557Z","shell.execute_reply":"2022-10-09T13:08:37.36613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4 color='blue' > There are 26K unique patients info available. Total numbers of records are 30K , but unqiue patien ID's are 26K . Looks there is some duplicated records for patient Id lets see.  </font>","metadata":{}},{"cell_type":"code","source":"sns.countplot(x='class',data=classInfodf);","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:37.370799Z","iopub.execute_input":"2022-10-09T13:08:37.372972Z","iopub.status.idle":"2022-10-09T13:08:37.638957Z","shell.execute_reply.started":"2022-10-09T13:08:37.372919Z","shell.execute_reply":"2022-10-09T13:08:37.63803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4 color='blue' > images from different classes \n1.Normal;\n2.No Lung Opacity / Not Normal; \n3.Lung Opacity </font>","metadata":{}},{"cell_type":"code","source":"def get_feature_distribution(data, feature):\n    # Get the count for each label\n    label_counts = data[feature].value_counts()\n\n    # Get total number of samples\n    total_samples = len(data)\n\n    # Count the number of items in each class\n    print(\"Feature: {}\".format(feature))\n    for i in range(len(label_counts)):\n        label = label_counts.index[i]\n        count = label_counts.values[i]\n        percent = int((count / total_samples) * 10000) / 100\n        print(\"{:<30s}:   {} or {}%\".format(label, count, percent))\n\nget_feature_distribution(classInfodf, 'class')","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:37.640273Z","iopub.execute_input":"2022-10-09T13:08:37.640522Z","iopub.status.idle":"2022-10-09T13:08:37.67286Z","shell.execute_reply.started":"2022-10-09T13:08:37.640476Z","shell.execute_reply":"2022-10-09T13:08:37.671898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classInfodf[classInfodf.duplicated()]","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:37.674196Z","iopub.execute_input":"2022-10-09T13:08:37.674445Z","iopub.status.idle":"2022-10-09T13:08:37.749686Z","shell.execute_reply.started":"2022-10-09T13:08:37.674398Z","shell.execute_reply":"2022-10-09T13:08:37.748943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4 color='blue' > As we suspected there are  3543 duplicated records . </font>","metadata":{}},{"cell_type":"code","source":"classInfodf[classInfodf.duplicated()].shape","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:37.753585Z","iopub.execute_input":"2022-10-09T13:08:37.755836Z","iopub.status.idle":"2022-10-09T13:08:37.781686Z","shell.execute_reply.started":"2022-10-09T13:08:37.755785Z","shell.execute_reply":"2022-10-09T13:08:37.780947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classInfodf[classInfodf.patientId=='c1f7889a-9ea9-4acb-b64c-b737c929599a']","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:37.785582Z","iopub.execute_input":"2022-10-09T13:08:37.787779Z","iopub.status.idle":"2022-10-09T13:08:37.811654Z","shell.execute_reply.started":"2022-10-09T13:08:37.78773Z","shell.execute_reply":"2022-10-09T13:08:37.810909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Check for missing values ","metadata":{}},{"cell_type":"code","source":"def missing_check(df):\n    total = df.isnull().sum().sort_values(ascending=False)  # total number of null values\n    percent = (df.isnull().sum() / df.isnull().count()).sort_values(\n        ascending=False)  # percentage of values that are null\n    missing_data = pd.concat([total, percent], axis=1, keys=['Total', 'Percent'])  # putting the above two together\n    return missing_data  # return the dataframe","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:37.815478Z","iopub.execute_input":"2022-10-09T13:08:37.817698Z","iopub.status.idle":"2022-10-09T13:08:37.825027Z","shell.execute_reply.started":"2022-10-09T13:08:37.817645Z","shell.execute_reply":"2022-10-09T13:08:37.824165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_check(classInfodf)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:37.829641Z","iopub.execute_input":"2022-10-09T13:08:37.832303Z","iopub.status.idle":"2022-10-09T13:08:37.907032Z","shell.execute_reply.started":"2022-10-09T13:08:37.832253Z","shell.execute_reply":"2022-10-09T13:08:37.906115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4 color='blue' > There are no missing values. </font>","metadata":{}},{"cell_type":"code","source":"trainlabelsdf.info()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:37.911107Z","iopub.execute_input":"2022-10-09T13:08:37.913393Z","iopub.status.idle":"2022-10-09T13:08:37.935428Z","shell.execute_reply.started":"2022-10-09T13:08:37.913334Z","shell.execute_reply":"2022-10-09T13:08:37.934724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4 color='blue' > Train labels also contains 30K records same as meta data . </font>","metadata":{}},{"cell_type":"code","source":"trainlabelsdf.patientId.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:37.939103Z","iopub.execute_input":"2022-10-09T13:08:37.941093Z","iopub.status.idle":"2022-10-09T13:08:37.961999Z","shell.execute_reply.started":"2022-10-09T13:08:37.941041Z","shell.execute_reply":"2022-10-09T13:08:37.961358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4 color='blue' > Unique patient Id count also matched with class info . </font>","metadata":{}},{"cell_type":"markdown","source":"<font size=4 color='blue' > Lets check for same patient which we verified the class Info.</font>","metadata":{}},{"cell_type":"code","source":"trainlabelsdf[trainlabelsdf.patientId=='c1f7889a-9ea9-4acb-b64c-b737c929599a']","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:37.965602Z","iopub.execute_input":"2022-10-09T13:08:37.967603Z","iopub.status.idle":"2022-10-09T13:08:38.000516Z","shell.execute_reply.started":"2022-10-09T13:08:37.967553Z","shell.execute_reply":"2022-10-09T13:08:37.999824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4 color='blue' > There are two records exists for same patient  but with different dimension. This may be due to in the X-ray opacity has been deteched at multiple locations. </font>","metadata":{}},{"cell_type":"code","source":"trainlabelsdf[trainlabelsdf.duplicated()]","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:38.004101Z","iopub.execute_input":"2022-10-09T13:08:38.006142Z","iopub.status.idle":"2022-10-09T13:08:38.038959Z","shell.execute_reply.started":"2022-10-09T13:08:38.006087Z","shell.execute_reply":"2022-10-09T13:08:38.038231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4 color='blue' > There are no duplicate records for train labels. </font>","metadata":{}},{"cell_type":"code","source":"sns.countplot(x='Target',data=trainlabelsdf);","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:38.042698Z","iopub.execute_input":"2022-10-09T13:08:38.044822Z","iopub.status.idle":"2022-10-09T13:08:38.35986Z","shell.execute_reply.started":"2022-10-09T13:08:38.044769Z","shell.execute_reply":"2022-10-09T13:08:38.358998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4 color='blue' > There are 2 values in Target column , those are either 1 or 0.  But  in classInfo we could see there are 3 different classes. Lets proceed further to check how Target and Class columns are related. </font>","metadata":{}},{"cell_type":"code","source":"print(f'No of entries which has Pneumonia: {trainlabelsdf[trainlabelsdf.Target == 1].shape[0]} i.e., {round(trainlabelsdf[trainlabelsdf.Target == 1].shape[0]/trainlabelsdf.shape[0]*100, 0)}%')\nprint(f'No of entries which don\\'t have Pneumonia: {trainlabelsdf[trainlabelsdf.Target == 0].shape[0]} i.e., {round(trainlabelsdf[trainlabelsdf.Target == 0].shape[0]/trainlabelsdf.shape[0]*100, 0)}%')\n_ = trainlabelsdf['Target'].value_counts().plot(kind = 'pie', autopct = '%.0f%%', labels = ['Negative', 'Positive'], figsize = (8 ,8))","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:38.361397Z","iopub.execute_input":"2022-10-09T13:08:38.361778Z","iopub.status.idle":"2022-10-09T13:08:38.623125Z","shell.execute_reply.started":"2022-10-09T13:08:38.36167Z","shell.execute_reply":"2022-10-09T13:08:38.622387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_check(trainlabelsdf)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:38.627188Z","iopub.execute_input":"2022-10-09T13:08:38.62922Z","iopub.status.idle":"2022-10-09T13:08:38.712129Z","shell.execute_reply.started":"2022-10-09T13:08:38.629162Z","shell.execute_reply":"2022-10-09T13:08:38.711366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainlabelsdf[trainlabelsdf.x.isna()]","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:38.716102Z","iopub.execute_input":"2022-10-09T13:08:38.718258Z","iopub.status.idle":"2022-10-09T13:08:38.776662Z","shell.execute_reply.started":"2022-10-09T13:08:38.718203Z","shell.execute_reply":"2022-10-09T13:08:38.775828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4 color='blue' > There are X,Y values aremissing  for few records . This can be due to the fact that for a normal patient these values could be not applicable.  </font>","metadata":{}},{"cell_type":"code","source":"trainlabelsdf[trainlabelsdf.x.isna()]['Target'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:38.78062Z","iopub.execute_input":"2022-10-09T13:08:38.782844Z","iopub.status.idle":"2022-10-09T13:08:38.797833Z","shell.execute_reply.started":"2022-10-09T13:08:38.782779Z","shell.execute_reply":"2022-10-09T13:08:38.797117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4 color='blue' > Its proved that only for normal patients dimensions are not available. </font>","metadata":{}},{"cell_type":"code","source":"trainlabelsdf.describe()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:38.801675Z","iopub.execute_input":"2022-10-09T13:08:38.803782Z","iopub.status.idle":"2022-10-09T13:08:38.863985Z","shell.execute_reply.started":"2022-10-09T13:08:38.803727Z","shell.execute_reply":"2022-10-09T13:08:38.863291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4 color='blue' > Looks , 75% of the data represents target value 1. Count plot also shows the same.  So looks its an imbalanced data set .  </font>","metadata":{}},{"cell_type":"markdown","source":"<font size=4 color='blue' > Lets concatenate classInfo and trainlabels . Before concatinating them lets remove duplicate records from class info. </font>","metadata":{}},{"cell_type":"code","source":"classInfodf.drop_duplicates(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:38.867817Z","iopub.execute_input":"2022-10-09T13:08:38.869897Z","iopub.status.idle":"2022-10-09T13:08:38.893692Z","shell.execute_reply.started":"2022-10-09T13:08:38.869824Z","shell.execute_reply":"2022-10-09T13:08:38.893006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 2: Map training and testing images to its classes. ","metadata":{}},{"cell_type":"code","source":"#traindf = pd.concat([classInfo,trainlabels]);\ntraindf = trainlabelsdf.merge(classInfodf, left_on='patientId', right_on='patientId', how='inner')\n","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:38.897614Z","iopub.execute_input":"2022-10-09T13:08:38.899739Z","iopub.status.idle":"2022-10-09T13:08:38.931951Z","shell.execute_reply.started":"2022-10-09T13:08:38.899683Z","shell.execute_reply":"2022-10-09T13:08:38.931215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traindf.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:38.935902Z","iopub.execute_input":"2022-10-09T13:08:38.938078Z","iopub.status.idle":"2022-10-09T13:08:38.971106Z","shell.execute_reply.started":"2022-10-09T13:08:38.938024Z","shell.execute_reply":"2022-10-09T13:08:38.970386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4 color='blue' > Lets check for specific  patient how data has been concatinated. </font>","metadata":{}},{"cell_type":"code","source":"traindf[traindf.patientId=='c1f7889a-9ea9-4acb-b64c-b737c929599a']","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:38.974914Z","iopub.execute_input":"2022-10-09T13:08:38.977065Z","iopub.status.idle":"2022-10-09T13:08:39.010157Z","shell.execute_reply.started":"2022-10-09T13:08:38.977014Z","shell.execute_reply":"2022-10-09T13:08:39.008705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4 color='blue' > Good, same patient has different dimension values with same target value.  </font>\n    \n   Now lets see <font size=4 color='orange' >  how class and target are related to each other?? </font>","metadata":{}},{"cell_type":"code","source":"traindf.Target.unique()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:39.011434Z","iopub.execute_input":"2022-10-09T13:08:39.012001Z","iopub.status.idle":"2022-10-09T13:08:39.018127Z","shell.execute_reply.started":"2022-10-09T13:08:39.011949Z","shell.execute_reply":"2022-10-09T13:08:39.017359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traindf.groupby(['class', 'Target']).size().reset_index(name='Patient Count')","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:39.019409Z","iopub.execute_input":"2022-10-09T13:08:39.019982Z","iopub.status.idle":"2022-10-09T13:08:39.050635Z","shell.execute_reply.started":"2022-10-09T13:08:39.01986Z","shell.execute_reply":"2022-10-09T13:08:39.049862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows=1,figsize=(12,6))\ntmp = traindf.groupby('Target')['class'].value_counts()\ndf = pd.DataFrame(data={'test': tmp.values}, index=tmp.index).reset_index()\nsns.barplot(ax=ax,x = 'Target', y='test',hue='class',data=df, palette='Set3')\nplt.title(\" class and Target\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:39.054803Z","iopub.execute_input":"2022-10-09T13:08:39.057095Z","iopub.status.idle":"2022-10-09T13:08:39.413732Z","shell.execute_reply.started":"2022-10-09T13:08:39.057037Z","shell.execute_reply":"2022-10-09T13:08:39.412802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4 color='blue' > Here is the answer , class with Normal and No Lung Opacity / Not Normal has been classified into single target  value that is 'O'. So we can summarize that the prediction which we have to do is like Patient has Lung Opacity ir not . Because Normal and not normal patients are combines in same Target .  </font>","metadata":{}},{"cell_type":"markdown","source":"<font size=4 color='blue'> Prediction: Binary classification  i.e. Patient has Lung Opacity or not ?   </font>","metadata":{}},{"cell_type":"code","source":"sns.boxplot(data=traindf)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:39.415065Z","iopub.execute_input":"2022-10-09T13:08:39.415576Z","iopub.status.idle":"2022-10-09T13:08:39.783312Z","shell.execute_reply.started":"2022-10-09T13:08:39.415329Z","shell.execute_reply":"2022-10-09T13:08:39.782272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traindf.corr()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:39.788836Z","iopub.execute_input":"2022-10-09T13:08:39.791673Z","iopub.status.idle":"2022-10-09T13:08:39.838425Z","shell.execute_reply.started":"2022-10-09T13:08:39.791597Z","shell.execute_reply":"2022-10-09T13:08:39.837778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target1 = traindf[traindf['Target']==1]\nsns.set_style('whitegrid')\nplt.figure()\nfig, ax = plt.subplots(2,2,figsize=(12,12))\nsns.distplot(target1['x'],kde=True,bins=50, color=\"red\", ax=ax[0,0]);\nsns.distplot(target1['y'],kde=True,bins=50, color=\"blue\", ax=ax[0,1]);\nsns.distplot(target1['width'],kde=True,bins=50, color=\"green\", ax=ax[1,0]);\nsns.distplot(target1['height'],kde=True,bins=50, color=\"magenta\", ax=ax[1,1]);\nlocs, labels = plt.xticks()\nplt.tick_params(axis='both', which='major', labelsize=12)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:39.841888Z","iopub.execute_input":"2022-10-09T13:08:39.843706Z","iopub.status.idle":"2022-10-09T13:08:41.047949Z","shell.execute_reply.started":"2022-10-09T13:08:39.843656Z","shell.execute_reply":"2022-10-09T13:08:41.047137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nprint(\"Number of images in train set:\", len(image_train_path),\"\\nNumber of images in test set:\", len(image_test_path))","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.048921Z","iopub.execute_input":"2022-10-09T13:08:41.049189Z","iopub.status.idle":"2022-10-09T13:08:41.055429Z","shell.execute_reply.started":"2022-10-09T13:08:41.049143Z","shell.execute_reply":"2022-10-09T13:08:41.054583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4 color='blue'> Train images length matched with unique patient id's in traindf. </font>","metadata":{}},{"cell_type":"markdown","source":"## Step 3: Map training and testing images to its annotations. ","metadata":{}},{"cell_type":"code","source":"samplePatientID = list(traindf[:3].T.to_dict().values())[0]['patientId']\ndcm_path = trainImagesPath/samplePatientID\ndcm_path = dcm_path.with_suffix(\".dcm\")\ndcm = pydicom.read_file(str(dcm_path))\ndcm","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:22:40.411657Z","iopub.execute_input":"2022-10-09T14:22:40.412022Z","iopub.status.idle":"2022-10-09T14:22:40.434196Z","shell.execute_reply.started":"2022-10-09T14:22:40.411953Z","shell.execute_reply":"2022-10-09T14:22:40.43341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4 color='blue'> We can observe that we do have available some useful information in the DICOM metadata with predictive value, for example: <br> </font>\n<font size=4 color='blue'>\nPatient sex;<br>\nPatient age;<br>\nModality;<br>\nBody part examined;<br>\nView position;<br>\nRows & Columns;<br>\nPixel Spacing. </font>","metadata":{}},{"cell_type":"code","source":"def show_dicom_images(data):\n    img_data = list(data.T.to_dict().values())\n    f, ax = plt.subplots(3,3, figsize=(16,18))\n    for i,data_row in enumerate(img_data):\n        dcm_path = trainImagesPath/data_row['patientId']\n        dcm_path = dcm_path.with_suffix(\".dcm\")\n        data_row_img_data = pydicom.read_file(str(dcm_path))\n        modality = data_row_img_data.Modality\n        age = data_row_img_data.PatientAge\n        sex = data_row_img_data.PatientSex\n        ax[i//3, i%3].imshow(data_row_img_data.pixel_array, cmap=plt.cm.bone) \n        ax[i//3, i%3].axis('off')\n        ax[i//3, i%3].set_title('ID: {}\\nModality: {} Age: {} Sex: {} Target: {}\\nClass: {}\\nWindow: {}:{}:{}:{}'.format(\n                data_row['patientId'],\n                modality, age, sex, data_row['Target'], data_row['class'], \n                data_row['x'],data_row['y'],data_row['width'],data_row['height']))\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:23:11.934294Z","iopub.execute_input":"2022-10-09T14:23:11.934593Z","iopub.status.idle":"2022-10-09T14:23:11.941131Z","shell.execute_reply.started":"2022-10-09T14:23:11.934538Z","shell.execute_reply":"2022-10-09T14:23:11.940314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_dicom_images(traindf[traindf['Target']==1].sample(9))","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:23:14.479529Z","iopub.execute_input":"2022-10-09T14:23:14.479823Z","iopub.status.idle":"2022-10-09T14:23:16.019686Z","shell.execute_reply.started":"2022-10-09T14:23:14.479765Z","shell.execute_reply":"2022-10-09T14:23:16.019024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4 color='blue'>  We would like to represent the images with the overlay boxes superposed. For this, we will need first to parse the whole dataset with Target = 1 and gather all coordinates of the windows showing a Lung Opacity on the same image. </font>","metadata":{}},{"cell_type":"code","source":"from tqdm.notebook import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:23:21.450851Z","iopub.execute_input":"2022-10-09T14:23:21.451203Z","iopub.status.idle":"2022-10-09T14:23:21.474083Z","shell.execute_reply.started":"2022-10-09T14:23:21.451141Z","shell.execute_reply":"2022-10-09T14:23:21.472441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_HEIGHT = 256\nIMAGE_WIDTH = 256\n\ndef resize_normalize(imagePath):\n    data_row_img_data = pydicom.read_file(imagePath)\n    img=data_row_img_data.pixel_array\n    img_resized = cv2.resize(img, (IMAGE_HEIGHT, IMAGE_WIDTH))\n    img_normalized = img_resized / float(255)\n    return img_normalized","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:23:23.719208Z","iopub.execute_input":"2022-10-09T14:23:23.719488Z","iopub.status.idle":"2022-10-09T14:23:23.724473Z","shell.execute_reply.started":"2022-10-09T14:23:23.719436Z","shell.execute_reply":"2022-10-09T14:23:23.72362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vars = ['Modality', 'PatientAge', 'PatientSex', 'BodyPartExamined', 'ViewPosition', 'ConversionType', 'Rows', 'Columns', 'PixelSpacing','imagePath','image']\n\ndef process_dicom_data(data_df,imagesPath):\n    print(data_df.shape)\n    for var in vars:\n        data_df[var] = None\n    image_names = os.listdir(imagesPath)\n    i=0\n    for img_name in image_names:\n        dcm_path = imagesPath/img_name\n        dcm_path = dcm_path.with_suffix(\".dcm\")\n        data_row_img_data = pydicom.read_file(str(dcm_path))\n        idx = (data_df['patientId']==data_row_img_data.PatientID)\n        data_df.loc[idx,'Modality'] = data_row_img_data.Modality\n        data_df.loc[idx,'PatientAge'] = pd.to_numeric(data_row_img_data.PatientAge)\n        data_df.loc[idx,'PatientSex'] = data_row_img_data.PatientSex\n        data_df.loc[idx,'BodyPartExamined'] = data_row_img_data.BodyPartExamined\n        data_df.loc[idx,'ViewPosition'] = data_row_img_data.ViewPosition\n        data_df.loc[idx,'ConversionType'] = data_row_img_data.ConversionType\n        data_df.loc[idx,'Rows'] = data_row_img_data.Rows\n        data_df.loc[idx,'Columns'] = data_row_img_data.Columns  \n        data_df.loc[idx,'PixelSpacing'] = str.format(\"{:4.3f}\",data_row_img_data.PixelSpacing[0])\n        data_df.loc[idx,'imagePath'] =dcm_path\n        \n       ","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:24:00.050068Z","iopub.execute_input":"2022-10-09T14:24:00.05036Z","iopub.status.idle":"2022-10-09T14:24:00.057857Z","shell.execute_reply.started":"2022-10-09T14:24:00.050303Z","shell.execute_reply":"2022-10-09T14:24:00.056981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nSAMPLE_SIZE = 2000","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:32:32.087699Z","iopub.execute_input":"2022-10-09T14:32:32.088015Z","iopub.status.idle":"2022-10-09T14:32:32.091602Z","shell.execute_reply.started":"2022-10-09T14:32:32.087953Z","shell.execute_reply":"2022-10-09T14:32:32.090791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traindf=traindf.sample(SAMPLE_SIZE)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:32:33.729841Z","iopub.execute_input":"2022-10-09T14:32:33.730197Z","iopub.status.idle":"2022-10-09T14:32:33.737294Z","shell.execute_reply.started":"2022-10-09T14:32:33.730133Z","shell.execute_reply":"2022-10-09T14:32:33.736468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"process_dicom_data(traindf,trainImagesPath)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:32:35.420934Z","iopub.execute_input":"2022-10-09T14:32:35.421246Z","iopub.status.idle":"2022-10-09T14:36:07.445257Z","shell.execute_reply.started":"2022-10-09T14:32:35.421185Z","shell.execute_reply":"2022-10-09T14:36:07.44401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traindf['image'] =  traindf['imagePath'].apply(resize_normalize)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:32:15.011873Z","iopub.status.idle":"2022-10-09T14:32:15.012329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\npath = os.getcwd()\ndf_pickle_file = os.path.join(path, 'dataframe_6k.pkl')\ntraindf.to_pickle(df_pickle_file)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:32:15.013225Z","iopub.status.idle":"2022-10-09T14:32:15.013995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testdf = pd.read_csv(sampleSubPath)\n\ntestdf = testdf.drop('PredictionString',1)\nprocess_dicom_data(testdf,testImagesPath)\n\ndf_test_pickle_file = os.path.join(path, 'test_6k.pkl')\ntestdf.to_pickle(df_test_pickle_file)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:32:15.014892Z","iopub.status.idle":"2022-10-09T14:32:15.015648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 4: Preprocessing and Visualisation of different classes ","metadata":{}},{"cell_type":"code","source":"traindf.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:32:15.01659Z","iopub.status.idle":"2022-10-09T14:32:15.017344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The meaning of this modality is CR - Computer Radiography. We can drop this column as its unique for all records.","metadata":{}},{"cell_type":"code","source":"testdf.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:32:15.018221Z","iopub.status.idle":"2022-10-09T14:32:15.01993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Patient Age","metadata":{}},{"cell_type":"code","source":"fig, (ax) = plt.subplots(nrows=1,figsize=(16,6))\nsns.countplot(ax=ax, x = 'PatientAge',hue='class',data=traindf, order = traindf['PatientAge'].value_counts().index)\nplt.title(\"Train set: Age and Class\")\nplt.xticks(rotation=90)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:32:15.020742Z","iopub.status.idle":"2022-10-09T14:32:15.02141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax) = plt.subplots(nrows=1,figsize=(16,6))\nsns.countplot(ax=ax, x = 'PatientAge',hue='Target',data=traindf, order = traindf['PatientAge'].value_counts().index)\nplt.title(\"Train set: Age and Target\")\nplt.xticks(rotation=90)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:32:15.022216Z","iopub.status.idle":"2022-10-09T14:32:15.022859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most of the data has been captured for age group between 40 to 50. There is an outlier with age 151. There are very few data points for age group between 1 to 5 and 80 to 90.","metadata":{}},{"cell_type":"code","source":"sns.boxplot(data=traindf, x='class', y='PatientAge');","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:32:15.023657Z","iopub.status.idle":"2022-10-09T14:32:15.02432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Note: The age range of class 'No Lung Opacity / Not Normal' seems to be different.\n","metadata":{}},{"cell_type":"markdown","source":"#### PatientSex","metadata":{}},{"cell_type":"code","source":"print('classes distribution for all patients:')\n_ = traindf['class'].value_counts().plot(kind = 'pie', autopct = '%.0f%%', figsize = (4, 4))","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:32:15.025146Z","iopub.status.idle":"2022-10-09T14:32:15.025805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('classes distribution for Female gender:')\n_ = traindf[traindf['PatientSex']=='F']['class'].value_counts().plot(kind = 'pie', autopct = '%.0f%%', figsize = (4, 4))","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:32:15.026624Z","iopub.status.idle":"2022-10-09T14:32:15.027287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('classes distribution for Male gender:')\n_ = traindf[traindf['PatientSex']=='M']['class'].value_counts().plot(kind = 'pie', autopct = '%.0f%%', figsize = (4, 4))","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:32:15.028152Z","iopub.status.idle":"2022-10-09T14:32:15.028833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In test set also similar kind of behaviour observed in data among different age groups. Outlier with Age 412 . ","metadata":{}},{"cell_type":"code","source":"traindf['PatientSex'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:32:15.029929Z","iopub.status.idle":"2022-10-09T14:32:15.030578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most of the data points for Male gender both in traing and testing set .","metadata":{}},{"cell_type":"code","source":"testdf['PatientSex'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:32:15.031434Z","iopub.status.idle":"2022-10-09T14:32:15.0321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax) = plt.subplots(nrows=1,figsize=(10,5))\nsns.countplot(ax=ax, x = 'PatientSex',hue='Target',data=traindf)\nplt.title(\"Train set: Gender and Target\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:32:15.032956Z","iopub.status.idle":"2022-10-09T14:32:15.03366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax) = plt.subplots(nrows=1,figsize=(10,5))\nsns.countplot(ax=ax, x = 'PatientSex',data=traindf)\nplt.title(\"Train set: Gender and Target\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:32:15.034459Z","iopub.status.idle":"2022-10-09T14:32:15.035101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax) = plt.subplots(nrows=1,figsize=(10,5))\nsns.countplot(ax=ax, x = 'PatientSex',data=testdf)\nplt.title(\"Test set: Gender and Target\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.326512Z","iopub.status.idle":"2022-10-09T13:08:41.327135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### BodyPartExamined,\n#### ConversionType\t\n#### Rows\tColumns","metadata":{}},{"cell_type":"markdown","source":"Unique values found for this column, we can drop.","metadata":{}},{"cell_type":"code","source":"traindf['Rows'].value_counts(),traindf['Columns'].value_counts(),","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.327827Z","iopub.status.idle":"2022-10-09T13:08:41.328428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Unique values found.drop it.","metadata":{}},{"cell_type":"markdown","source":"#### ViewPosition","metadata":{}},{"cell_type":"code","source":"traindf['ViewPosition'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.329172Z","iopub.status.idle":"2022-10-09T13:08:41.32975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='ViewPosition',hue='Target',data=traindf)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.330481Z","iopub.status.idle":"2022-10-09T13:08:41.331086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='ViewPosition',hue='class',data=traindf)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.331805Z","iopub.status.idle":"2022-10-09T13:08:41.332412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import preprocessing\n  \n# label_encoder object knows how to understand word labels.\nlabel_encoder = preprocessing.LabelEncoder()\n  \ntraindf['ViewPosition']= label_encoder.fit_transform(traindf['ViewPosition'])\nprint(label_encoder.classes_)\ntraindf['ViewPosition'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.333158Z","iopub.status.idle":"2022-10-09T13:08:41.333747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='ViewPosition',hue='class',data=traindf)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.334471Z","iopub.status.idle":"2022-10-09T13:08:41.335071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"AP-AnteriorPosterior=0\nPA-PostereiorAterior=1","metadata":{}},{"cell_type":"markdown","source":"#### PixelSpacing","metadata":{}},{"cell_type":"code","source":"traindf['PixelSpacing'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.33579Z","iopub.status.idle":"2022-10-09T13:08:41.336463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='PixelSpacing',hue='class',data=traindf)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.337215Z","iopub.status.idle":"2022-10-09T13:08:41.337788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.scatterplot(traindf['PixelSpacing'] , traindf['PatientAge'], hue=traindf['Target'],palette=['green','red'])","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.338528Z","iopub.status.idle":"2022-10-09T13:08:41.339127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='PixelSpacing',hue='Target',data=traindf)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.339847Z","iopub.status.idle":"2022-10-09T13:08:41.340453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"0.115 Pixel spacing for agane between 2-11 years. and 0.199 is from 20-35 agae group. we have very less data points for this.","metadata":{}},{"cell_type":"code","source":"traindf[traindf['PixelSpacing']=='0.199']","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.341207Z","iopub.status.idle":"2022-10-09T13:08:41.341784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traindf[traindf['PixelSpacing']=='0.115']","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.34257Z","iopub.status.idle":"2022-10-09T13:08:41.34317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"PixelSpacing isvery specific to image related to feature the way it is captured , so it can be dropped.","metadata":{}},{"cell_type":"code","source":"traindf=traindf.drop(['Modality','ConversionType','BodyPartExamined','Rows','Columns','PixelSpacing'],1);","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.343897Z","iopub.status.idle":"2022-10-09T13:08:41.344486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traindf.info()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.345224Z","iopub.status.idle":"2022-10-09T13:08:41.345843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traindf[traindf['PatientAge'].isna()]","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.346611Z","iopub.status.idle":"2022-10-09T13:08:41.34722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def drawgraphs(data_file, columns, hue = False, width = 15, showdistribution = True):\n  if (hue):\n    print('Creating graph for: {} and {}'.format(columns, hue))\n  else:  \n    print('Creating graph for : {}'.format(columns))\n  length = len(columns) * 6\n  total = float(len(data_file))\n\n  fig, axes = plt.subplots(nrows = len(columns) if len(columns) > 1 else 1, ncols = 1, figsize = (width, length))\n  for index, content in enumerate(columns):\n    plt.title(content)\n\n    currentaxes = 0\n    if (len(columns) > 1):\n      currentaxes = axes[index]\n    else:\n      currentaxes = axes\n\n    if (hue):\n      sns.countplot(x = columns[index], data = data_file, ax = currentaxes, hue = hue)\n    else:\n      sns.countplot(x = columns[index], data = data_file, ax = currentaxes)\n\n    if(showdistribution):\n      for p in (currentaxes.patches):\n        height = p.get_height()\n        if (height > 0 and total > 0):\n          currentaxes.text(p.get_x() + p.get_width()/2., height + 3, '{:1.2f}%'.format(100*height/total), ha = \"center\")","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.347961Z","iopub.status.idle":"2022-10-09T13:08:41.348556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drawgraphs(data_file = traindf, columns = ['PatientSex'], hue = False, width = 10, showdistribution = True)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.349286Z","iopub.status.idle":"2022-10-09T13:08:41.349886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drawgraphs(data_file = traindf, columns = ['PatientSex'], hue = 'Target', showdistribution = True)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.350612Z","iopub.status.idle":"2022-10-09T13:08:41.35123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"age_25 = np.percentile(traindf['PatientAge'], 25)\nage_75 = np.percentile(traindf['PatientAge'], 75)\niqr_age = age_75 - age_25\ncutoff_age = 1.5 * iqr_age\n\nlow_lim_age = age_25 - cutoff_age\nupp_lim_age = age_75 + cutoff_age\n\noutlier_age = [x for x in traindf['PatientAge'] if x < low_lim_age or x > upp_lim_age]\nprint('The number of outliers in `PatientAge` out of 30277 records are: ', len(outlier_age))\nsns.boxplot(traindf['PatientAge'], orient = 'h').set_title('Outliers in PatientAge')\nprint('\\nThe ages which are in the outlier categories are:', outlier_age)\n\nfig = plt.figure(figsize = (10, 6))","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.351958Z","iopub.status.idle":"2022-10-09T13:08:41.352545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Distribution of `PatientAge`: Overall and Target = 1')\nfig = plt.figure(figsize = (10, 6))\n\nax = fig.add_subplot(121)\ng = (sns.distplot(traindf['PatientAge']).set_title('Distribution of PatientAge'))\n\nax = fig.add_subplot(122)\ng = (sns.distplot(traindf.loc[traindf['Target'] == 1, 'PatientAge']).set_title('Distribution of PatientAge vs PnuemoniaEvidence'))","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.353284Z","iopub.status.idle":"2022-10-09T13:08:41.353956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### ","metadata":{}},{"cell_type":"markdown","source":"## Step 5: Display images with bounding box.","metadata":{}},{"cell_type":"code","source":"def show_dicom_images_with_boxes(data):\n    img_data = list(data.T.to_dict().values())\n    f, ax = plt.subplots(3,3, figsize=(16,18))\n    for i,data_row in enumerate(img_data):\n        dcm_path = trainImagesPath/data_row['patientId']\n        dcm_path = dcm_path.with_suffix(\".dcm\")\n        data_row_img_data = pydicom.read_file(dcm_path)\n        modality = data_row_img_data.Modality\n        age = data_row_img_data.PatientAge\n        sex = data_row_img_data.PatientSex\n        ax[i//3, i%3].imshow(data_row_img_data.pixel_array, cmap=plt.cm.bone) \n        ax[i//3, i%3].axis('off')\n        ax[i//3, i%3].set_title('ID: {}\\nModality: {} Age: {} Sex: {} Target: {}\\nClass: {}'.format(\n                data_row['patientId'],modality, age, sex, data_row['Target'], data_row['class']))\n        rows = traindf[traindf['patientId']==data_row['patientId']]\n        box_data = list(rows.T.to_dict().values())\n        for j, row in enumerate(box_data):\n            ax[i//3, i%3].add_patch(Rectangle(xy=(row['x'], row['y']),\n                        width=row['width'],height=row['height'], \n                       linewidth=2, edgecolor='r', facecolor='none'))   \n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.354706Z","iopub.status.idle":"2022-10-09T13:08:41.355309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_dicom_images_with_boxes(traindf[traindf['Target']==1].sample(9))","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.356089Z","iopub.status.idle":"2022-10-09T13:08:41.356723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nshow_dicom_images_with_boxes(traindf[traindf['Target']==0].sample(9))","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.357441Z","iopub.status.idle":"2022-10-09T13:08:41.358045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 6: Design, train and test basic CNN models for classification.","metadata":{}},{"cell_type":"code","source":"traindf.info()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.358755Z","iopub.status.idle":"2022-10-09T13:08:41.359365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntraindf['class'].replace(['No Lung Opacity / Not Normal','Lung Opacity','Normal'],[0,1,2],inplace=True);","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.360101Z","iopub.status.idle":"2022-10-09T13:08:41.360669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#SAMPLE_SIZE = 2000\nprint(traindf['Target'].unique())\n#traindf_smpl = traindf[traindf['class'] == 'Normal'].sample(math.ceil(SAMPLE_SIZE / 3.0))\n#traindf_smpl = traindf_smpl.append(traindf[traindf['class'] == 'Lung Opacity'].sample(math.ceil(SAMPLE_SIZE / 3.0)))\n#traindf_smpl = traindf_smpl.append(traindf[traindf['class'] == 'No Lung Opacity / Not Normal'].sample(math.ceil(SAMPLE_SIZE / 3.0)))\n#traindf_smpl=traindf","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.361414Z","iopub.status.idle":"2022-10-09T13:08:41.362025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.362738Z","iopub.status.idle":"2022-10-09T13:08:41.36335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_HEIGHT = 256\nIMAGE_WIDTH = 256\n\nX = traindf['image']\ny = traindf['Target']\nX_arr = []\nfor img in X:\n    X_arr.append(np.array(np.reshape(img, (IMAGE_HEIGHT, IMAGE_WIDTH, 1))))\nX_arr = np.array(X_arr, np.float32)\n\nprint(X_arr.shape)\nprint(len(y))","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.36408Z","iopub.status.idle":"2022-10-09T13:08:41.364654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X\ngc.collect()\n\nX = X_arr\nprint('X image data: ', X.shape)\nprint('individual image: ', X[0].shape)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.365394Z","iopub.status.idle":"2022-10-09T13:08:41.366022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ny_enc =  traindf[\"Target\"];\n\nprint('y_enc shape: ', y_enc.shape)\nX_train, X_test, y_train, y_test = train_test_split(X, y_enc, test_size=0.30, random_state=42)\n\nprint('X_train: ', X_train.shape)\nprint('y_train: ', y_train.shape)\nprint('X_test:  ', X_test.shape)\nprint('y_test:  ', y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.36673Z","iopub.status.idle":"2022-10-09T13:08:41.367324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#Print metrics and return score for test and train in Data frame format.   \ndef printMetrics(model,X_train, X_test, y_train, y_test):\n    y_train_pred = model.predict(X_train)\n    y_pred = model.predict(X_test)\n    report=classification_report(y_train,y_train_pred);\n    \n    print(\"\\n----Classification report for Train data------ \\n \"+report)\n    accuracy=accuracy_score(y_train,y_train_pred)\n    lines = report.split('\\n');\n    report_data = []\n    for line in lines[2:-5]:\n        row = {}\n        row_data = line.split('      ')\n        row['Target'] = row_data[1]\n        row['precision'] = float(row_data[2])\n        row['recall'] = float(row_data[3])\n        row['f1_score'] = float(row_data[4])\n        row['Accuracy']=float(accuracy)\n        report_data.append(row)\n    modelTrainScore = pd.DataFrame.from_dict(report_data)\n    print(\"\\n----Confusion Matrix for Train data------ \\n \")\n    ax=sns.heatmap(confusion_matrix(y_train, y_train_pred), annot=True, fmt='.2f');\n    ax.set_xlabel('\\nPredicted Values')\n    ax.set_ylabel('Actual Values ');\n    plt.show();\n    accuracy=accuracy_score(y_test,y_pred)\n\n    report=classification_report(y_test,y_pred);\n    print(\"\\n----Classification report for Test data------ \\n \"+report)\n    lines = report.split('\\n');\n    report_data = []\n    for line in lines[2:-5]:\n        row = {}\n        row_data = line.split('      ')\n        row['Target'] = row_data[1]\n        row['precision'] = float(row_data[2])\n        row['recall'] = float(row_data[3])\n        row['f1_score'] = float(row_data[4])\n        row['Accuracy']=float(accuracy)\n        report_data.append(row)\n    modelTestScore = pd.DataFrame.from_dict(report_data)\n\n    print(\"\\n----Confusion Matrix for Test data------ \\n \")\n    ax=sns.heatmap(confusion_matrix(y_test, y_pred), annot=True, fmt='.2f');\n    ax.set_xlabel('\\nPredicted Values')\n    ax.set_ylabel('Actual Values ');\n    plt.show(); \n       \n    return modelTrainScore,modelTestScore;\n    \n\nfrom numpy import argmax\nfrom numpy import sqrt\ndef plot_roc_pre(model,X_test,y_test):\n    y_proba = model.predict_proba(X_test)\n    y_true = np.argmax(y_test, axis=0)\n    roc_auc_test = roc_auc_score(y_true, y_proba[:,1],multi_class=\"ovr\")\n    fpr, tpr, thresholds = roc_curve(y_true, y_proba[:,1])\n    # calculate the g-mean for each threshold\n    gmeans = sqrt(tpr * (1-fpr))\n    # locate the index of the largest g-mean\n    ix = argmax(gmeans)\n    plt.figure(figsize=(15,5))\n    plt.subplot(1,2,1)\n    # plot no skill\n    plt.plot([0, 1], [0, 1],'r--',label='No Skill')\n    plt.scatter(fpr[ix], tpr[ix], marker='o', color='black', label='Best')\n    # plot the roc curve for the model\n    plt.plot(fpr, tpr, marker='.',label=type(model).__name__+'(area = %0.2f)' % roc_auc_test)\n    plt.title(\"ROC curve\")\n    plt.xlabel('false positive rate')\n    plt.ylabel('true positive rate')\n    plt.legend(loc=\"lower right\")\n    plt.subplot(1,2,2)\n    precision, recall, thresholds = precision_recall_curve(y_true, y_proba[:,1])\n    # convert to f score\n    fscore = (2 * precision * recall) / (precision + recall)\n    # locate the index of the largest f score\n    ix = argmax(fscore)\n    print('Best Threshold=%f, F-Score=%.3f' % (thresholds[ix], fscore[ix]))\n    plt.plot([0, 1], [0.5, 0.5], linestyle='--')\n    plt.scatter(recall[ix], precision[ix], marker='o', color='black', label='Best')\n\n    # plot the precision-recall curve for the model\n    plt.plot(recall, precision, marker='.',label=type(model).__name__+'(area = %0.2f)' % roc_auc_test)\n    \n    plt.title(\"precision recall curve\")\n    plt.xlabel('Recall')\n    plt.ylabel('Precision')\n    # show the plot\n    plt.show()\n     ","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.368132Z","iopub.status.idle":"2022-10-09T13:08:41.368718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"nsamples = X_train.shape[0]\nnx =  X_train.shape[1]\nny =  X_train.shape[2]\nd2_train_dataset = X_train.reshape((nsamples,nx*ny))\nnsamples = X_test.shape[0]\nnx =  X_test.shape[1]\nny =  X_test.shape[2]\nd2_test_dataset = X_test.reshape((nsamples,nx*ny))\n\n\n\nlogreg = LogisticRegression()\nlogreg.fit(d2_train_dataset, y_train)\n#Y_pred = logreg.predict(X_valdiation)\nacc_log = round(logreg.score(d2_train_dataset, y_train) * 100, 4)\nacc_log\n#X_valdiation, y_validation","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.369455Z","iopub.status.idle":"2022-10-09T13:08:41.370062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"printMetrics(logreg,d2_train_dataset, d2_test_dataset, y_train, y_test)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.370782Z","iopub.status.idle":"2022-10-09T13:08:41.37139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### CNN Model","metadata":{}},{"cell_type":"code","source":"\nclass_enc = LabelEncoder()\nclasses_encoded = class_enc.fit_transform(y)\nprint('classes encoded: ', classes_encoded)\nprint('min: %d, max: %d' % (min(classes_encoded), max(classes_encoded)))\nprint('size: ', len(classes_encoded))\nprint(class_enc.classes_)\n\ny_enc = to_categorical(classes_encoded)\nprint('y_enc shape: ', y_enc.shape)\nX_train, X_test, y_train, y_test = train_test_split(X, y_enc, test_size=0.30, random_state=42)\n\nprint('X_train: ', X_train.shape)\nprint('y_train: ', y_train.shape)\nprint('X_test:  ', X_test.shape)\nprint('y_test:  ', y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.372135Z","iopub.status.idle":"2022-10-09T13:08:41.372722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DROPOUT_PERCENT = 0.4\n\ndef create_model(learning_rate, verbose=False):\n    print('\\ncreating model with learning rate: ', learning_rate)\n\n    # Initialising the CNN classifier\n    classifier = Sequential()\n\n    # Add a Convolution layer with 32 kernels of 3X3 shape with activation function ReLU\n    classifier.add(Conv2D(32, (3, 3), input_shape = (IMAGE_HEIGHT, IMAGE_WIDTH, 1), \n                          activation = 'relu', padding = 'same'))\n\n    # Add a Max Pooling layer of size 2X2\n    classifier.add(MaxPooling2D(pool_size = (2, 2)))\n\n    # Add another Convolution layer with 32 kernels of 3X3 shape with activation function ReLU\n    classifier.add(Conv2D(32, (3, 3), activation = 'relu', padding = 'same'))\n\n    # Adding another pooling layer\n    classifier.add(MaxPooling2D(pool_size = (2, 2)))\n\n    # Add another Convolution layer with 32 kernels of 3X3 shape with activation function ReLU\n    classifier.add(Conv2D(32, (3, 3), activation = 'relu', padding = 'same'))\n\n    # Adding another pooling layer\n    classifier.add(MaxPooling2D(pool_size = (2, 2)))\n\n    # Flattening the layer before fully connected layers\n    classifier.add(Flatten())\n\n    # Adding a fully connected layer with 512 neurons\n    classifier.add(Dense(units = 512, activation = 'relu'))\n\n    # Adding dropout with probability defined above\n    classifier.add(Dropout(DROPOUT_PERCENT))\n\n    # Adding a fully connected layer with 128 neurons\n    classifier.add(Dense(units = 128, activation = 'relu'))\n\n    # The final output layer with 3 neuron to predict the categorical classifcation\n    classifier.add(Dense(units = 3, activation = 'softmax'))\n    \n    opt = Adam(learning_rate=learning_rate, beta_1=0.9, beta_2=0.999, epsilon=None, decay=0.001, amsgrad=False)\n    classifier.compile(optimizer = opt, loss = 'categorical_crossentropy', metrics = ['accuracy'])\n    \n    return classifier\n","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.373491Z","iopub.status.idle":"2022-10-09T13:08:41.374092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train\n\ndef train_model(model, X_img_train, y_spc_train, batch_size, epochs, verbose):\n    history = model.fit(X_img_train, y_spc_train, batch_size=batch_size, epochs=epochs, \n                        validation_split=0.3, initial_epoch=0)\n    return history","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.374803Z","iopub.status.idle":"2022-10-09T13:08:41.375405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test\n\ndef print_classification_report(model, X_img_test, y_cls_test):\n    print('classification report:')\n    y_test = np.argmax(y_cls_test, axis=1)\n    Y_pred = model.predict(X_img_test)\n    y_pred = np.argmax(Y_pred, axis=1)    \n    report = classification_report(y_test, y_pred)\n    print(report)\n\ndef evaluate_model(model, X_img_test, y_cls_test, verb):\n    score = model.evaluate(X_img_test, y_cls_test, batch_size=batch_size, verbose=verb)\n    return score","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.376221Z","iopub.status.idle":"2022-10-09T13:08:41.376867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_and_test(X_img_train, y_spc_train, X_img_test, y_spc_test, batch_size, epochs, learning_rate, verbose):\n    model = create_model(learning_rate, verbose)\n    history = train_model(model, X_img_train, y_spc_train, batch_size, epochs, verbose)\n    score = evaluate_model(model, X_img_test, y_spc_test, True)\n    print('\\nloss on test data: %s, accuracy on test data: %0.2f%%' % (score[0], score[1]*100.0))\n    return model, history","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.377609Z","iopub.status.idle":"2022-10-09T13:08:41.378207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learning_rate = 0.001\nbatch_size = 128\nepochs = 10\nmodel, history = train_and_test(X_train, y_train, X_test, y_test, batch_size, epochs, learning_rate, True)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.378929Z","iopub.status.idle":"2022-10-09T13:08:41.37951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('X_train: ', X_train.shape)\nprint('y_train: ', y_train.shape)\nprint('X_test:  ', X_test.shape)\nprint('y_test:  ', y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.380257Z","iopub.status.idle":"2022-10-09T13:08:41.380841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Pre Processing the image\nfrom tensorflow.keras.applications.mobilenet import preprocess_input\nimport matplotlib.pyplot as plt\nimport pydicom\nfrom pydicom.data import get_testdata_files\n\nimages = []\nADJUSTED_IMAGE_SIZE = 128\nimageList = []\nclassLabels = []\nlabels = []\noriginalImage = []\n# Function to read the image from the path and reshape the image to size\ndef readAndReshapeImage(image):\n    img = np.array(image).astype(np.uint8)\n    ## Resize the image\n    res = cv2.resize(img,(ADJUSTED_IMAGE_SIZE,ADJUSTED_IMAGE_SIZE), interpolation = cv2.INTER_LINEAR)\n    return res\n\n## Read the imahge and resize the image\ndef populateImage(rowData):\n    for index, row in rowData.iterrows():\n        patientId = row.patientId\n        classlabel = row[\"class\"]\n        dcm_file = 'stage_2_train_images/'+'{}.dcm'.format(patientId)\n        #print(dcm_file)\n        #dcm_data = dcm.read_file(dcm_file)\n        dcm_data = pydicom.dcmread(dcm_file)\n        #print(dcm_data)\n        img = dcm_data.pixel_array\n        ## Converting the image to 3 channels as the dicom image pixel does not have colour classes wiht it\n        # Refer the link https://stackoverflow.com/questions/40119743/convert-a-grayscale-image-to-a-3-channel-image\n        if len(img.shape) != 3 or img.shape[2] != 3:\n            img = np.stack((img,) * 3, -1)\n        imageList.append(readAndReshapeImage(img))\n        classLabels.append(classlabel)\n    tmpImages = np.array(imageList)\n    tmpLabels = np.array(classLabels)\n    return tmpImages,tmpLabels","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.381594Z","iopub.status.idle":"2022-10-09T13:08:41.382189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Reading the images into numpy array\nimages,labels = populateImage(traindf)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.382914Z","iopub.status.idle":"2022-10-09T13:08:41.383519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.384243Z","iopub.status.idle":"2022-10-09T13:08:41.384817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.385579Z","iopub.status.idle":"2022-10-09T13:08:41.386207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelBinarizer\nenc = LabelBinarizer()\ny2 = enc.fit_transform(labels)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.386976Z","iopub.status.idle":"2022-10-09T13:08:41.38755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(images, y2, test_size=0.3, random_state=50)\nX_test, X_val, y_test, y_val = train_test_split(X_test,y_test, test_size = 0.5, random_state=50)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.38828Z","iopub.status.idle":"2022-10-09T13:08:41.388857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications.vgg16 import VGG16\nfrom tensorflow.keras.applications.vgg16 import preprocess_input\n\n#base_model = VGG16(weights=\"imagenet\", include_top=False, input_shape=X_train[0].shape)\nbase_model = VGG16(input_shape=X_train[0].shape, weights=None, include_top=False)\nbase_model.load_weights('vgg16_weights_tf_dim_ordering_tf_kernels_notop.h5')\nbase_model.trainable = False ## Not trainable weights\n\n## Preprocessing input\ntrain_ds = preprocess_input(X_train) \ntrain_val_df = preprocess_input(X_val)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.38963Z","iopub.status.idle":"2022-10-09T13:08:41.390229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras import layers, models\n\nflatten_layer = layers.Flatten()\ndense_layer_1 = layers.Dense(50, activation='relu')\ndense_layer_2 = layers.Dense(20, activation='relu')\nprediction_layer = layers.Dense(3, activation='softmax')\n\n\ncnn_VGG16_model = models.Sequential([\n    base_model,\n    flatten_layer,\n    dense_layer_1,\n    dense_layer_2,\n    prediction_layer\n])","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.390957Z","iopub.status.idle":"2022-10-09T13:08:41.391549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping\n\ncnn_VGG16_model.compile(\n    optimizer='Adam',\n    loss=tensorflow.keras.losses.BinaryCrossentropy(from_logits=True),\n    metrics=['accuracy'],\n)\n\n## Early stopping whe validation accuracy does not change for 7 iteration \nes = EarlyStopping(monitor='val_accuracy', mode='auto', patience=7,  restore_best_weights=True)\n\n#Trainign the model\nhistory = cnn_VGG16_model.fit(train_ds, y_train, epochs=10, validation_data=(train_val_df,y_val) ,callbacks=es)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.392308Z","iopub.status.idle":"2022-10-09T13:08:41.392913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications.resnet50 import ResNet50 \nfrom tensorflow.keras.applications.resnet50 import preprocess_input\nfrom tensorflow.keras import layers, models\n\nresnet_base_model = ResNet50(include_top=False, weights=None, input_shape=X_train[0].shape)\nresnet_base_model.load_weights('resnet50_weights_tf_dim_ordering_tf_kernels_notop.h5')\n\ntrain_ds = preprocess_input(X_train) \ntrain_val_df = preprocess_input(X_val)\n\n\nflatten_layer = layers.Flatten()\ndense_layer_1 = layers.Dense(50, activation='relu')\ndense_layer_2 = layers.Dense(32, activation='relu')\nprediction_layer = layers.Dense(3, activation='softmax')\n\ncnn_resnet_model = models.Sequential([\n    resnet_base_model,\n    flatten_layer,\n    dense_layer_1,\n    dense_layer_2,\n    prediction_layer\n])","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.393645Z","iopub.status.idle":"2022-10-09T13:08:41.394246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping\n\ncnn_resnet_model.compile(\n    optimizer='Adam',\n    loss=tensorflow.keras.losses.BinaryCrossentropy(from_logits=True),\n    metrics=['accuracy'],\n)\nhistory =cnn_resnet_model.fit(train_ds, y_train, epochs=10, validation_data=(train_val_df,y_val))","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.395084Z","iopub.status.idle":"2022-10-09T13:08:41.395726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"weights_mobilenet_v3_large_minimalistic_224_1.0_float_no_top.h5\nweights_mobilenet_v3_small_224_1.0_float_no_top.h5\nweights_mobilenet_v3_small_minimalistic_224_1.0_float_no_top.h5\nweights_mobilenet_v3_large_224_1.0_float_no_top.h5\nmobilenet_v2_weights_tf_dim_ordering_tf_kernels_1.0_192_no_top\nmobilenet_1_0_192_tf_no_top.h5\nNASNet-mobile-no-top.h5","metadata":{}},{"cell_type":"code","source":"from keras.layers import Flatten, GlobalAveragePooling2D,GlobalMaxPooling2D\nfrom tensorflow.keras.optimizers import RMSprop\ndef cnn_model(height, width, num_channels, num_classes, loss='categorical_crossentropy', metrics=['accuracy']):\n  batch_size = None\n\n  model = Sequential()\n\n  model.add(Conv2D(filters = 32, kernel_size = (5,5),padding = 'Same', \n                  activation ='relu', batch_input_shape = (batch_size,height, width, num_channels)))\n\n\n  model.add(Conv2D(filters = 32, kernel_size = (5,5),padding = 'Same', \n                  activation ='relu'))\n  model.add(MaxPooling2D(pool_size=(2,2)))\n # model.add(Dropout(0.2))\n\n\n  model.add(Conv2D(filters = 64, kernel_size = (3,3),padding = 'Same', \n                  activation ='relu'))\n  model.add(Conv2D(filters = 64, kernel_size = (3,3),padding = 'same', \n                  activation ='relu'))\n  model.add(MaxPooling2D(pool_size=(2,2), strides=(2,2)))\n  #model.add(Dropout(0.3))\n\n  model.add(Conv2D(filters = 128, kernel_size = (3,3),padding = 'Same', \n                  activation ='relu'))\n  model.add(Conv2D(filters = 128, kernel_size = (3,3),padding = 'Same', \n                  activation ='relu'))\n  model.add(MaxPooling2D(pool_size=(2,2), strides=(2,2)))\n  #model.add(Dropout(0.4))\n\n\n\n  model.add(GlobalMaxPooling2D())\n  model.add(Dense(256, activation = \"relu\"))\n  #model.add(Dropout(0.5))\n  model.add(Dense(num_classes, activation = \"softmax\"))\n\n  optimizer = RMSprop(lr=0.001, rho=0.9, epsilon=1e-08, decay=0.0)\n  model.compile(optimizer = optimizer, loss = loss, metrics = metrics)\n  model.summary()\n  return model","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.39657Z","iopub.status.idle":"2022-10-09T13:08:41.397202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cnn = cnn_model(ADJUSTED_IMAGE_SIZE,ADJUSTED_IMAGE_SIZE,3,3)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.397928Z","iopub.status.idle":"2022-10-09T13:08:41.398519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = cnn.fit(X_train, \n                  y_train, \n                  epochs = 10,\n                  validation_data = (X_val,y_val),\n                  batch_size = 30)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.399264Z","iopub.status.idle":"2022-10-09T13:08:41.399839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.400574Z","iopub.status.idle":"2022-10-09T13:08:41.401188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fcl_loss, fcl_accuracy = cnn.evaluate(X_test, y_test, verbose=1)\nprint('Test loss:', fcl_loss)\nprint('Test accuracy:', fcl_accuracy)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:08:41.401918Z","iopub.status.idle":"2022-10-09T13:08:41.402509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_DIR = '/kaggle/input'\n\n# Directory to save logs and trained model\nROOT_DIR = '/kaggle/working'","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:09:58.597485Z","iopub.execute_input":"2022-10-09T13:09:58.597801Z","iopub.status.idle":"2022-10-09T13:09:58.60278Z","shell.execute_reply.started":"2022-10-09T13:09:58.597744Z","shell.execute_reply":"2022-10-09T13:09:58.601395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!git clone https://www.github.com/matterport/Mask_RCNN.git\nos.chdir('Mask_RCNN')","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:10:01.153667Z","iopub.execute_input":"2022-10-09T13:10:01.154131Z","iopub.status.idle":"2022-10-09T13:10:02.30041Z","shell.execute_reply.started":"2022-10-09T13:10:01.154073Z","shell.execute_reply":"2022-10-09T13:10:02.298818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python setup.py -q install","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:10:02.448142Z","iopub.execute_input":"2022-10-09T13:10:02.448416Z","iopub.status.idle":"2022-10-09T13:10:06.553942Z","shell.execute_reply.started":"2022-10-09T13:10:02.448361Z","shell.execute_reply":"2022-10-09T13:10:06.553124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nsys.path.append(os.path.join(ROOT_DIR, 'Mask_RCNN'))  # To find local version of the library\nfrom mrcnn.config import Config\nfrom mrcnn import utils\nimport mrcnn.model as modellib\nfrom mrcnn import visualize\nfrom mrcnn.model import log","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:10:06.557051Z","iopub.execute_input":"2022-10-09T13:10:06.557326Z","iopub.status.idle":"2022-10-09T13:10:06.707779Z","shell.execute_reply.started":"2022-10-09T13:10:06.557284Z","shell.execute_reply":"2022-10-09T13:10:06.70617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# bonding box identification","metadata":{}},{"cell_type":"code","source":"DATA_DIR = '/kaggle/input'\n\n# Directory to save logs and trained model\nROOT_DIR = '/kaggle/working'","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:10:08.935432Z","iopub.execute_input":"2022-10-09T13:10:08.935738Z","iopub.status.idle":"2022-10-09T13:10:08.939924Z","shell.execute_reply.started":"2022-10-09T13:10:08.935679Z","shell.execute_reply":"2022-10-09T13:10:08.938754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nsys.path.append(os.path.join(ROOT_DIR, 'Mask_RCNN'))  # To find local version of the library\nfrom mrcnn.config import Config\nfrom mrcnn import utils\nimport mrcnn.model as modellib\nfrom mrcnn import visualize\nfrom mrcnn.model import log","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:10:10.433814Z","iopub.execute_input":"2022-10-09T13:10:10.434173Z","iopub.status.idle":"2022-10-09T13:10:10.439166Z","shell.execute_reply.started":"2022-10-09T13:10:10.434109Z","shell.execute_reply":"2022-10-09T13:10:10.4382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dicom_dir = os.path.join(DATA_DIR, 'stage_2_train_images')\ntest_dicom_dir = os.path.join(DATA_DIR, 'stage_2_test_images')","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:10:12.413614Z","iopub.execute_input":"2022-10-09T13:10:12.413995Z","iopub.status.idle":"2022-10-09T13:10:12.418089Z","shell.execute_reply.started":"2022-10-09T13:10:12.413925Z","shell.execute_reply":"2022-10-09T13:10:12.4173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dicom_dir","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:10:17.532126Z","iopub.execute_input":"2022-10-09T13:10:17.532419Z","iopub.status.idle":"2022-10-09T13:10:17.540859Z","shell.execute_reply.started":"2022-10-09T13:10:17.532363Z","shell.execute_reply":"2022-10-09T13:10:17.540012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_dicom_fps(dicom_dir):\n    dicom_fps = glob.glob(dicom_dir+'/'+'*.dcm')\n    return list(set(dicom_fps))\n\ndef parse_dataset(dicom_dir, anns): \n    image_fps = get_dicom_fps(dicom_dir)\n    image_annotations = {fp: [] for fp in image_fps}\n    for index, row in anns.iterrows(): \n        fp = os.path.join(dicom_dir, row['patientId']+'.dcm')\n        image_annotations[fp].append(row)\n    return image_fps, image_annotations ","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:10:19.402082Z","iopub.execute_input":"2022-10-09T13:10:19.402384Z","iopub.status.idle":"2022-10-09T13:10:19.408107Z","shell.execute_reply.started":"2022-10-09T13:10:19.402329Z","shell.execute_reply":"2022-10-09T13:10:19.406955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The following parameters have been selected to reduce running time for demonstration purposes \n# These are not optimal \n\nclass DetectorConfig(Config):\n    \"\"\"Configuration for training pneumonia detection on the RSNA pneumonia dataset.\n    Overrides values in the base Config class.\n    \"\"\"\n    \n    # Give the configuration a recognizable name  \n    NAME = 'pneumonia'\n    \n    # Train on 1 GPU and 8 images per GPU. We can put multiple images on each\n    # GPU because the images are small. Batch size is 8 (GPUs * images/GPU).\n    GPU_COUNT = 1\n    IMAGES_PER_GPU = 8\n    \n    BACKBONE = 'resnet50'\n    \n    NUM_CLASSES = 2  # background + 1 pneumonia classes\n    \n    IMAGE_MIN_DIM = 256\n    IMAGE_MAX_DIM = 256\n    RPN_ANCHOR_SCALES = (32, 64, 128, 256)\n    TRAIN_ROIS_PER_IMAGE = 32\n    MAX_GT_INSTANCES = 3\n    DETECTION_MAX_INSTANCES = 3\n    DETECTION_MIN_CONFIDENCE = 0.7\n    DETECTION_NMS_THRESHOLD = 0.1\n\n    STEPS_PER_EPOCH = 200\n    \nconfig = DetectorConfig()\nconfig.display()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:10:23.091019Z","iopub.execute_input":"2022-10-09T13:10:23.091313Z","iopub.status.idle":"2022-10-09T13:10:23.106233Z","shell.execute_reply.started":"2022-10-09T13:10:23.091255Z","shell.execute_reply":"2022-10-09T13:10:23.105107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DetectorDataset(utils.Dataset):\n    \"\"\"Dataset class for training pneumonia detection on the RSNA pneumonia dataset.\n    \"\"\"\n\n    def __init__(self, image_fps, image_annotations, orig_height, orig_width):\n        super().__init__(self)\n        \n        # Add classes\n        self.add_class('pneumonia', 1, 'Lung Opacity')\n        \n        # add images \n        for i, fp in enumerate(image_fps):\n            annotations = image_annotations[fp]\n            self.add_image('pneumonia', image_id=i, path=fp, \n                           annotations=annotations, orig_height=orig_height, orig_width=orig_width)\n            \n    def image_reference(self, image_id):\n        info = self.image_info[image_id]\n        return info['path']\n\n    def load_image(self, image_id):\n        info = self.image_info[image_id]\n        fp = info['path']\n        ds = pydicom.read_file(fp)\n        image = ds.pixel_array\n        # If grayscale. Convert to RGB for consistency.\n        if len(image.shape) != 3 or image.shape[2] != 3:\n            image = np.stack((image,) * 3, -1)\n        return image\n\n    def load_mask(self, image_id):\n        info = self.image_info[image_id]\n        annotations = info['annotations']\n        count = len(annotations)\n        if count == 0:\n            mask = np.zeros((info['orig_height'], info['orig_width'], 1), dtype=np.uint8)\n            class_ids = np.zeros((1,), dtype=np.int32)\n        else:\n            mask = np.zeros((info['orig_height'], info['orig_width'], count), dtype=np.uint8)\n            class_ids = np.zeros((count,), dtype=np.int32)\n            for i, a in enumerate(annotations):\n                if a['Target'] == 1:\n                    x = int(a['x'])\n                    y = int(a['y'])\n                    w = int(a['width'])\n                    h = int(a['height'])\n                    mask_instance = mask[:, :, i].copy()\n                    cv2.rectangle(mask_instance, (x, y), (x+w, y+h), 255, -1)\n                    mask[:, :, i] = mask_instance\n                    class_ids[i] = 1\n        return mask.astype(np.bool), class_ids.astype(np.int32)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:10:26.937565Z","iopub.execute_input":"2022-10-09T13:10:26.937913Z","iopub.status.idle":"2022-10-09T13:10:26.949316Z","shell.execute_reply.started":"2022-10-09T13:10:26.937832Z","shell.execute_reply":"2022-10-09T13:10:26.948372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training dataset\nDATA_DIR = '/kaggle/input/'\nanns = pd.read_csv(os.path.join(DATA_DIR, 'stage_2_train_labels.csv'))\nanns.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:10:30.268739Z","iopub.execute_input":"2022-10-09T13:10:30.269074Z","iopub.status.idle":"2022-10-09T13:10:30.327345Z","shell.execute_reply.started":"2022-10-09T13:10:30.269011Z","shell.execute_reply":"2022-10-09T13:10:30.326398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nimage_fps, image_annotations = parse_dataset(train_dicom_dir, anns=anns)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:10:32.563057Z","iopub.execute_input":"2022-10-09T13:10:32.56364Z","iopub.status.idle":"2022-10-09T13:10:35.315305Z","shell.execute_reply.started":"2022-10-09T13:10:32.563575Z","shell.execute_reply":"2022-10-09T13:10:35.314503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds = pydicom.read_file(image_fps[0]) # read dicom image from filepath \nimage = ds.pixel_array # get image array","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:10:35.316519Z","iopub.execute_input":"2022-10-09T13:10:35.316787Z","iopub.status.idle":"2022-10-09T13:10:35.355069Z","shell.execute_reply.started":"2022-10-09T13:10:35.316743Z","shell.execute_reply":"2022-10-09T13:10:35.354328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:10:35.819707Z","iopub.execute_input":"2022-10-09T13:10:35.820207Z","iopub.status.idle":"2022-10-09T13:10:35.832544Z","shell.execute_reply.started":"2022-10-09T13:10:35.820002Z","shell.execute_reply":"2022-10-09T13:10:35.831528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Original DICOM image size: 1024 x 1024\nORIG_SIZE = 1024","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:10:39.167124Z","iopub.execute_input":"2022-10-09T13:10:39.167448Z","iopub.status.idle":"2022-10-09T13:10:39.17137Z","shell.execute_reply.started":"2022-10-09T13:10:39.167393Z","shell.execute_reply":"2022-10-09T13:10:39.170612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# split dataset into training vs. validation dataset \nimage_fps_list = list(image_fps)\nrandom.seed(42)\nrandom.shuffle(image_fps_list)\n\nval_size = 1500\nimage_fps_val = image_fps_list[:val_size]\nimage_fps_train = image_fps_list[val_size:]\n\nprint(len(image_fps_train), len(image_fps_val))","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:10:41.53857Z","iopub.execute_input":"2022-10-09T13:10:41.538878Z","iopub.status.idle":"2022-10-09T13:10:41.5735Z","shell.execute_reply.started":"2022-10-09T13:10:41.538808Z","shell.execute_reply":"2022-10-09T13:10:41.571984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prepare the training dataset\ndataset_train = DetectorDataset(image_fps_train, image_annotations, ORIG_SIZE, ORIG_SIZE)\ndataset_train.prepare()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:10:43.722477Z","iopub.execute_input":"2022-10-09T13:10:43.722768Z","iopub.status.idle":"2022-10-09T13:10:43.795405Z","shell.execute_reply.started":"2022-10-09T13:10:43.722712Z","shell.execute_reply":"2022-10-09T13:10:43.794596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show annotation(s) for a DICOM image \ntest_fp = random.choice(image_fps_train)\nimage_annotations[test_fp]","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:10:46.830719Z","iopub.execute_input":"2022-10-09T13:10:46.831026Z","iopub.status.idle":"2022-10-09T13:10:46.849516Z","shell.execute_reply.started":"2022-10-09T13:10:46.83097Z","shell.execute_reply":"2022-10-09T13:10:46.847958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prepare the validation dataset\ndataset_val = DetectorDataset(image_fps_val, image_annotations, ORIG_SIZE, ORIG_SIZE)\ndataset_val.prepare()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:10:48.909541Z","iopub.execute_input":"2022-10-09T13:10:48.909832Z","iopub.status.idle":"2022-10-09T13:10:48.917917Z","shell.execute_reply.started":"2022-10-09T13:10:48.909777Z","shell.execute_reply":"2022-10-09T13:10:48.917076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load and display random sample and their bounding boxes\n\nclass_ids = [0]\nwhile class_ids[0] == 0:  ## look for a mask\n    image_id = random.choice(dataset_train.image_ids)\n    image_fp = dataset_train.image_reference(image_id)\n    image = dataset_train.load_image(image_id)\n    mask, class_ids = dataset_train.load_mask(image_id)\n\nprint(image.shape)\n\nplt.figure(figsize=(10, 10))\nplt.subplot(1, 2, 1)\nplt.imshow(image)\nplt.axis('off')\n\nplt.subplot(1, 2, 2)\nmasked = np.zeros(image.shape[:2])\nfor i in range(mask.shape[2]):\n    masked += image[:, :, 0] * mask[:, :, i]\nplt.imshow(masked, cmap='gray')\nplt.axis('off')\n\nprint(image_fp)\nprint(class_ids)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:10:51.340827Z","iopub.execute_input":"2022-10-09T13:10:51.341199Z","iopub.status.idle":"2022-10-09T13:10:51.704787Z","shell.execute_reply.started":"2022-10-09T13:10:51.341138Z","shell.execute_reply":"2022-10-09T13:10:51.703998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image augmentation (light but constant)\nfrom imgaug import augmenters as iaa\naugmentation = iaa.Sequential([\n    iaa.OneOf([ ## geometric transform\n        iaa.Affine(\n            scale={\"x\": (0.98, 1.04), \"y\": (0.98, 1.04)},\n            translate_percent={\"x\": (-0.03, 0.03), \"y\": (-0.05, 0.05)},\n            rotate=(-5, 5),\n            shear=(-3, 3),\n        ),\n        iaa.PiecewiseAffine(scale=(0.002, 0.03)),\n    ]),\n    iaa.OneOf([ ## brightness or contrast\n        iaa.Multiply((0.85, 1.15)),\n        iaa.ContrastNormalization((0.85, 1.15)),\n    ]),\n    iaa.OneOf([ ## blur or sharpen\n        iaa.GaussianBlur(sigma=(0.0, 0.12)),\n        iaa.Sharpen(alpha=(0.0, 0.12)),\n    ]),\n])\n\n# test on the same image as above\nimggrid = augmentation.draw_grid(image[:, :, 0], cols=5, rows=2)\nplt.figure(figsize=(30, 12))\n_ = plt.imshow(imggrid[:, :, 0], cmap='gray')","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:10:55.299745Z","iopub.execute_input":"2022-10-09T13:10:55.300097Z","iopub.status.idle":"2022-10-09T13:11:00.81271Z","shell.execute_reply.started":"2022-10-09T13:10:55.30003Z","shell.execute_reply":"2022-10-09T13:11:00.811963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#ROOT_DIR = path\nmodel = modellib.MaskRCNN(mode='training', config=config, model_dir=ROOT_DIR)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:11:00.813829Z","iopub.execute_input":"2022-10-09T13:11:00.814272Z","iopub.status.idle":"2022-10-09T13:11:07.77621Z","shell.execute_reply.started":"2022-10-09T13:11:00.814225Z","shell.execute_reply":"2022-10-09T13:11:07.775528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUM_EPOCHS = 16\nLEARNING_RATE = 0.006\n\n# Train Mask-RCNN Model \nimport warnings \nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:11:07.777624Z","iopub.execute_input":"2022-10-09T13:11:07.77808Z","iopub.status.idle":"2022-10-09T13:11:07.782444Z","shell.execute_reply.started":"2022-10-09T13:11:07.778016Z","shell.execute_reply":"2022-10-09T13:11:07.781716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n## first epochs with higher lr to speedup the learning\nmodel.train(dataset_train, dataset_val,\n            learning_rate=LEARNING_RATE*2,\n            epochs=2,\n            layers='all',\n            augmentation=augmentation)  ## no need to augment yet","metadata":{"execution":{"iopub.status.busy":"2022-10-09T13:11:07.783563Z","iopub.execute_input":"2022-10-09T13:11:07.784301Z","iopub.status.idle":"2022-10-09T14:04:17.501999Z","shell.execute_reply.started":"2022-10-09T13:11:07.78419Z","shell.execute_reply":"2022-10-09T14:04:17.501043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmodel.train(dataset_train, dataset_val,\n            learning_rate=LEARNING_RATE,\n            epochs=NUM_EPOCHS,\n            layers='all',\n            augmentation=augmentation)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T14:36:17.174837Z","iopub.execute_input":"2022-10-09T14:36:17.175172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# select trained model \ndir_names = next(os.walk(model.model_dir))[1]\nkey = config.NAME.lower()\ndir_names = filter(lambda f: f.startswith(key), dir_names)\ndir_names = sorted(dir_names)\n\nif not dir_names:\n    import errno\n    raise FileNotFoundError(\n        errno.ENOENT,\n        \"Could not find model directory under {}\".format(self.model_dir))\n    \nfps = []\n# Pick last directory\nfor d in dir_names: \n    dir_name = os.path.join(model.model_dir, d)\n    # Find the last checkpoint\n    checkpoints = next(os.walk(dir_name))[2]\n    checkpoints = filter(lambda f: f.startswith(\"mask_rcnn\"), checkpoints)\n    checkpoints = sorted(checkpoints)\n    if not checkpoints:\n        print('No weight files in {}'.format(dir_name))\n    else:\n        checkpoint = os.path.join(dir_name, checkpoints[-1])\n        fps.append(checkpoint)\n\nmodel_path = sorted(fps)[-1]\nprint('Found model {}'.format(model_path))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class InferenceConfig(DetectorConfig):\n    GPU_COUNT = 1\n    IMAGES_PER_GPU = 1\n\ninference_config = InferenceConfig()\n\n# Recreate the model in inference mode\nmodel = modellib.MaskRCNN(mode='inference', \n                          config=inference_config,\n                          model_dir=ROOT_DIR)\n\n# Load trained weights (fill in path to trained weights here)\nassert model_path != \"\", \"Provide path to trained weights\"\nprint(\"Loading weights from \", model_path)\nmodel.load_weights(model_path, by_name=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# set color for class\ndef get_colors_for_class_ids(class_ids):\n    colors = []\n    for class_id in class_ids:\n        if class_id == 1:\n            colors.append((.941, .204, .204))\n    return colors","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show few example of ground truth vs. predictions on the validation dataset \ndataset = dataset_val\nfig = plt.figure(figsize=(10, 30))\n\nfor i in range(6):\n    image_id = random.choice(dataset.image_ids)\n    \n    original_image, image_meta, gt_class_id, gt_bbox, gt_mask = \\\n        modellib.load_image_gt(dataset_val, inference_config, \n                               image_id, use_mini_mask=False)\n    \n    print(original_image.shape)\n    plt.subplot(6, 2, 2*i + 1)\n    visualize.display_instances(original_image, gt_bbox, gt_mask, gt_class_id, \n                                dataset.class_names,\n                                colors=get_colors_for_class_ids(gt_class_id), ax=fig.axes[-1])\n    \n    plt.subplot(6, 2, 2*i + 2)\n    results = model.detect([original_image]) #, verbose=1)\n    r = results[0]\n    visualize.display_instances(original_image, r['rois'], r['masks'], r['class_ids'], \n                                dataset.class_names, r['scores'], \n                                colors=get_colors_for_class_ids(r['class_ids']), ax=fig.axes[-1])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get filenames of test dataset DICOM images\ntest_image_fps = get_dicom_fps(test_dicom_dir)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make predictions on test images, write out sample submission\ndef predict(image_fps, filepath='submission.csv', min_conf=0.98):\n    # assume square image\n    resize_factor = ORIG_SIZE / config.IMAGE_SHAPE[0]\n    #resize_factor = ORIG_SIZE\n    with open(filepath, 'w') as file:\n        file.write(\"patientId,PredictionString\\n\")\n\n        for image_id in tqdm(image_fps):\n            ds = pydicom.read_file(image_id)\n            image = ds.pixel_array\n            # If grayscale. Convert to RGB for consistency.\n            if len(image.shape) != 3 or image.shape[2] != 3:\n                image = np.stack((image,) * 3, -1)\n            image, window, scale, padding, crop = utils.resize_image(\n                image,\n                min_dim=config.IMAGE_MIN_DIM,\n                min_scale=config.IMAGE_MIN_SCALE,\n                max_dim=config.IMAGE_MAX_DIM,\n                mode=config.IMAGE_RESIZE_MODE)\n\n            patient_id = os.path.splitext(os.path.basename(image_id))[0]\n\n            results = model.detect([image])\n            r = results[0]\n\n            out_str = \"\"\n            out_str += patient_id\n            out_str += \",\"\n            assert( len(r['rois']) == len(r['class_ids']) == len(r['scores']) )\n            if len(r['rois']) == 0:\n                pass\n            else:\n                num_instances = len(r['rois'])\n\n                for i in range(num_instances):\n                    if r['scores'][i] > min_conf:\n                        out_str += ' '\n                        out_str += str(round(r['scores'][i], 2))\n                        out_str += ' '\n\n                        # x1, y1, width, height\n                        x1 = r['rois'][i][1]\n                        y1 = r['rois'][i][0]\n                        width = r['rois'][i][3] - x1\n                        height = r['rois'][i][2] - y1\n                        bboxes_str = \"{} {} {} {}\".format(x1*resize_factor, y1*resize_factor, \\\n                                                           width*resize_factor, height*resize_factor)\n                        out_str += bboxes_str\n\n            file.write(out_str+\"\\n\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_fp = os.path.join(ROOT_DIR, 'submission.csv')\npredict(test_image_fps, filepath=submission_fp)\nprint(submission_fp)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.read_csv(submission_fp, names=['patientId', 'PredictionString'])\noutput.head(60)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show a few test image detection example\ndef visualize(): \n    image_id = random.choice(test_image_fps)\n    ds = pydicom.read_file(image_id)\n    \n    # original image \n    image = ds.pixel_array\n    \n    # assume square image \n    resize_factor = ORIG_SIZE / config.IMAGE_SHAPE[0]\n    \n    # If grayscale. Convert to RGB for consistency.\n    if len(image.shape) != 3 or image.shape[2] != 3:\n        image = np.stack((image,) * 3, -1) \n    resized_image, window, scale, padding, crop = utils.resize_image(\n        image,\n        min_dim=config.IMAGE_MIN_DIM,\n        min_scale=config.IMAGE_MIN_SCALE,\n        max_dim=config.IMAGE_MAX_DIM,\n        mode=config.IMAGE_RESIZE_MODE)\n\n    patient_id = os.path.splitext(os.path.basename(image_id))[0]\n    print(patient_id)\n\n    results = model.detect([resized_image])\n    r = results[0]\n    for bbox in r['rois']: \n        print(bbox)\n        x1 = int(bbox[1] * resize_factor)\n        y1 = int(bbox[0] * resize_factor)\n        x2 = int(bbox[3] * resize_factor)\n        y2 = int(bbox[2]  * resize_factor)\n        cv2.rectangle(image, (x1,y1), (x2,y2), (77, 255, 9), 3, 1)\n        width = x2 - x1 \n        height = y2 - y1 \n        print(\"x {} y {} h {} w {}\".format(x1, y1, width, height))\n    plt.figure() \n    plt.imshow(image, cmap=plt.cm.gist_gray)\n\nvisualize()\nvisualize()\nvisualize()\nvisualize()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# remove files to allow committing (hit files limit otherwise)\n!rm -rf /kaggle/working/Mask_RCNN","metadata":{},"execution_count":null,"outputs":[]}]}