{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"## This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n        \n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-16T12:52:16.72958Z","iopub.execute_input":"2021-07-16T12:52:16.730243Z","iopub.status.idle":"2021-07-16T12:52:16.741408Z","shell.execute_reply.started":"2021-07-16T12:52:16.730146Z","shell.execute_reply":"2021-07-16T12:52:16.740446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\nimport numpy as np\nimport pandas as pd\n\nimport pydicom as dicom\nimport os\nimport cv2\nimport PIL # optional\nimport shutil\nimport tensorflow as tf\nimport shutil\nimport SimpleITK\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2021-07-16T12:52:16.742928Z","iopub.execute_input":"2021-07-16T12:52:16.743196Z","iopub.status.idle":"2021-07-16T12:52:23.600819Z","shell.execute_reply.started":"2021-07-16T12:52:16.74317Z","shell.execute_reply":"2021-07-16T12:52:23.59999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading input data file in png format to dataframes\n#train images in png format and stored in annotation format in dataframe\nimport os\nkf = pd.DataFrame(columns=['data_type','StudyInstanceUID','image_file_ID','image_file_path'])\nfor dirname, _, filenames in os.walk('/kaggle/input/siim-covid19-detection-dicom-to-png-conversion'):\n    for filename in filenames:\n        \n        std_id = filename.split('_')[0]\n        #f_name_id = filename.split('_')[1].split('.')[0]\n        dat_type = dirname.split('/')[-1]\n        img_path = os.path.join(dirname, filename)\n        \n        kf.loc[kf.shape[0]] = [dat_type,std_id,filename,img_path]\n\n\npng_data = pd.DataFrame(kf.drop([0,1,2,3,4],axis=0).values,columns=kf.columns)\npng_data['image_file_ID'] = png_data.image_file_ID.str[13:-4]\npng_data\n\n# creating train and test datframes where all the image_study and studyInstanceUID are stored\npng_test = png_data[png_data['data_type'] == 'test']\npng_train = png_data[png_data['data_type'] == 'train']","metadata":{"execution":{"iopub.status.busy":"2021-07-16T12:52:23.602491Z","iopub.execute_input":"2021-07-16T12:52:23.60274Z","iopub.status.idle":"2021-07-16T12:52:58.218317Z","shell.execute_reply.started":"2021-07-16T12:52:23.602717Z","shell.execute_reply":"2021-07-16T12:52:58.217256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = pd.read_csv(\"/kaggle/input/siim-covid19-detection/sample_submission.csv\")\ntrain_image = pd.read_csv(\"/kaggle/input/siim-covid19-detection/train_image_level.csv\")\ntrain_study = pd.read_csv(\"/kaggle/input/siim-covid19-detection/train_study_level.csv\")\n\n# class label colum and case study id_number columns are created\n\ntrain_study['classification_class'] = np.zeros((train_study.shape[0],1))\nfor i in range(0,train_study.shape[0]):\n    train_study['classification_class'][i] = train_study.iloc[i].values[1:-1]\n\n# StudyInstanceUID \ntrain_study['StudyInstanceUID'] = train_study.id.str[0:-6]\ntrain_study","metadata":{"execution":{"iopub.status.busy":"2021-07-16T12:52:58.219927Z","iopub.execute_input":"2021-07-16T12:52:58.220208Z","iopub.status.idle":"2021-07-16T12:53:02.215633Z","shell.execute_reply.started":"2021-07-16T12:52:58.220184Z","shell.execute_reply":"2021-07-16T12:53:02.214563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"itr = 0\ncl = []\nfor i in train_study['classification_class'].values:\n    if i[0]==1:\n        cl.append('Negative')\n    elif i[1]==1:\n        cl.append('Typical')\n    elif i[2]==1:\n        cl.append('Indeterminate')\n    elif i[3]==1:\n        cl.append('Atypical')\n    else:\n        a=1\n    itr = itr+1\n\ntrain_study['class_type'] = cl\ntrain_study","metadata":{"execution":{"iopub.status.busy":"2021-07-16T12:53:02.216978Z","iopub.execute_input":"2021-07-16T12:53:02.217316Z","iopub.status.idle":"2021-07-16T12:53:02.255645Z","shell.execute_reply.started":"2021-07-16T12:53:02.217285Z","shell.execute_reply":"2021-07-16T12:53:02.254638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This data frame is for the object localisation and classification combined activities\n\ntrain_Study_image = pd.merge(train_image,train_study,on='StudyInstanceUID',how=\"inner\")\n#\ntrain_Study_image.columns = ['Image_id', 'boxes', 'label', 'StudyInstanceUID', 'Study_id',\n       'Negative for Pneumonia', 'Typical Appearance',\n       'Indeterminate Appearance', 'Atypical Appearance',\n       'classification_class','class_type']\n\ntrain_Study_image","metadata":{"execution":{"iopub.status.busy":"2021-07-16T12:53:02.257004Z","iopub.execute_input":"2021-07-16T12:53:02.257424Z","iopub.status.idle":"2021-07-16T12:53:02.300288Z","shell.execute_reply.started":"2021-07-16T12:53:02.257381Z","shell.execute_reply":"2021-07-16T12:53:02.299268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"std_img_df = train_Study_image.dropna(subset=['boxes'],axis=0)\nstd_img_df","metadata":{"execution":{"iopub.status.busy":"2021-07-16T12:53:02.301951Z","iopub.execute_input":"2021-07-16T12:53:02.302409Z","iopub.status.idle":"2021-07-16T12:53:02.345815Z","shell.execute_reply.started":"2021-07-16T12:53:02.30237Z","shell.execute_reply":"2021-07-16T12:53:02.34496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dcm_df =pd.read_csv('/kaggle/input/dicom-df/Siim_Covid_dcm_df.csv')\ndcm_df.set_index('image_file_id',inplace=True)\ndcm_df","metadata":{"execution":{"iopub.status.busy":"2021-07-16T12:53:02.346989Z","iopub.execute_input":"2021-07-16T12:53:02.347264Z","iopub.status.idle":"2021-07-16T12:53:02.42844Z","shell.execute_reply.started":"2021-07-16T12:53:02.347241Z","shell.execute_reply":"2021-07-16T12:53:02.427568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def scale_factor(dcm_image_path):\n               \n    ds_array = SimpleITK.ReadImage(dcm_image_path)  \n    img_array = SimpleITK.GetArrayFromImage(ds_array)  \n    x_fct = 256 /img_array.shape[2] # divide by width\n    y_fct = 256 /img_array.shape[1] # divide by height\n    \n    return [x_fct,y_fct]","metadata":{"execution":{"iopub.status.busy":"2021-07-16T12:53:02.432404Z","iopub.execute_input":"2021-07-16T12:53:02.432652Z","iopub.status.idle":"2021-07-16T12:53:02.436645Z","shell.execute_reply.started":"2021-07-16T12:53:02.432629Z","shell.execute_reply":"2021-07-16T12:53:02.435994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ncols = ['Study_image_id','images_label','Case_label','width','height','Xmin','Ymin','Xmax','Ymax','image_file_path']\ntrn_csv = pd.DataFrame(columns=cols)\nitr = 0\n\nfor r in np.arange(std_img_df.shape[0]):\n    row = std_img_df.iloc[r]\n    n = int(len(row.label.split(' '))/6)\n    \n    dcm_img_path = dcm_df.loc[row.Image_id[0:-6]].image_file_path\n    scl_fctr = scale_factor(dcm_img_path)\n    \n    for i in np.arange(n):\n        k = i*6\n        box = row.label.split(' ')[k:k+6]\n        opac = box[0]\n        xmin = float(box[2])*scl_fctr[0]\n        ymin = float(box[3])*scl_fctr[1]\n        w_h = 256\n        xmax = xmin + (float(box[4]))*scl_fctr[0]\n        ymax = ymin + (float(box[5]))*scl_fctr[1]\n        \n        c_mid = row.StudyInstanceUID + '_' + row.Image_id[0:-6]\n        im_path = '/kaggle/input/siim-covid19-detection-dicom-to-png-conversion/siim-covid19-detection/train/' + c_mid + '.png'\n        case_label = row.class_type\n        trn_csv.loc[trn_csv.shape[0]] = [c_mid,opac,case_label,w_h,w_h,xmin,ymin,xmax,ymax,im_path]","metadata":{"execution":{"iopub.status.busy":"2021-07-16T12:53:02.4377Z","iopub.execute_input":"2021-07-16T12:53:02.437966Z","iopub.status.idle":"2021-07-16T13:25:50.76746Z","shell.execute_reply.started":"2021-07-16T12:53:02.437941Z","shell.execute_reply":"2021-07-16T13:25:50.758016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trn_csv.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-16T13:25:50.788642Z","iopub.execute_input":"2021-07-16T13:25:50.789054Z","iopub.status.idle":"2021-07-16T13:25:50.843973Z","shell.execute_reply.started":"2021-07-16T13:25:50.789005Z","shell.execute_reply":"2021-07-16T13:25:50.842941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr2 = trn_csv.drop('images_label',axis = 1)\ntr1 = trn_csv.drop('Case_label',axis = 1)","metadata":{"execution":{"iopub.status.busy":"2021-07-16T13:25:50.845155Z","iopub.execute_input":"2021-07-16T13:25:50.845419Z","iopub.status.idle":"2021-07-16T13:25:50.85992Z","shell.execute_reply.started":"2021-07-16T13:25:50.845393Z","shell.execute_reply":"2021-07-16T13:25:50.859141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr1.columns = ['Study_image_id', 'label', 'width', 'height', 'Xmin', 'Ymin',\n       'Xmax', 'Ymax', 'image_file_path']\n\ntr2.columns = ['Study_image_id', 'label', 'width', 'height', 'Xmin', 'Ymin',\n       'Xmax', 'Ymax', 'image_file_path']","metadata":{"execution":{"iopub.status.busy":"2021-07-16T13:25:50.861237Z","iopub.execute_input":"2021-07-16T13:25:50.861503Z","iopub.status.idle":"2021-07-16T13:25:50.868815Z","shell.execute_reply.started":"2021-07-16T13:25:50.861477Z","shell.execute_reply":"2021-07-16T13:25:50.867938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_final_csv = pd.concat([tr1,tr2],axis=0,ignore_index=True)\ntrain_final_csv","metadata":{"execution":{"iopub.status.busy":"2021-07-16T13:25:50.869881Z","iopub.execute_input":"2021-07-16T13:25:50.870187Z","iopub.status.idle":"2021-07-16T13:25:50.901706Z","shell.execute_reply.started":"2021-07-16T13:25:50.870162Z","shell.execute_reply":"2021-07-16T13:25:50.90079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_final_csv.to_csv('final_train_csv.csv')","metadata":{"execution":{"iopub.status.busy":"2021-07-16T13:25:50.903056Z","iopub.execute_input":"2021-07-16T13:25:50.903592Z","iopub.status.idle":"2021-07-16T13:25:51.190731Z","shell.execute_reply.started":"2021-07-16T13:25:50.903552Z","shell.execute_reply":"2021-07-16T13:25:51.189948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dft = pd.read_csv('/kaggle/input/siim-train-csv/final_train_csv.csv')\ndft.drop(['Unnamed: 0'],axis = 1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2021-07-16T13:25:51.19215Z","iopub.execute_input":"2021-07-16T13:25:51.192749Z","iopub.status.idle":"2021-07-16T13:25:51.282925Z","shell.execute_reply.started":"2021-07-16T13:25:51.192705Z","shell.execute_reply":"2021-07-16T13:25:51.282108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dft","metadata":{"execution":{"iopub.status.busy":"2021-07-16T13:25:51.284156Z","iopub.execute_input":"2021-07-16T13:25:51.284735Z","iopub.status.idle":"2021-07-16T13:25:51.30623Z","shell.execute_reply.started":"2021-07-16T13:25:51.284697Z","shell.execute_reply":"2021-07-16T13:25:51.305329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}