{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"## This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n        \n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-24T10:22:16.264918Z","iopub.execute_input":"2021-06-24T10:22:16.265374Z","iopub.status.idle":"2021-06-24T10:22:16.269874Z","shell.execute_reply.started":"2021-06-24T10:22:16.265332Z","shell.execute_reply":"2021-06-24T10:22:16.269063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\nimport numpy as np\nimport pandas as pd\n\nimport pydicom as dicom\nimport os\nimport cv2\nimport PIL # optional\nimport shutil","metadata":{"execution":{"iopub.status.busy":"2021-06-24T10:22:17.792840Z","iopub.execute_input":"2021-06-24T10:22:17.793462Z","iopub.status.idle":"2021-06-24T10:22:20.183764Z","shell.execute_reply.started":"2021-06-24T10:22:17.793423Z","shell.execute_reply":"2021-06-24T10:22:20.182904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading input data file in png format to dataframes\n#train images in png format and stored in annotation format in dataframe\nimport os\nkf = pd.DataFrame(columns=['data_type','StudyInstanceUID','image_file_ID','image_file_path'])\nfor dirname, _, filenames in os.walk('/kaggle/input/siim-covid19-detection-dicom-to-png-conversion'):\n    for filename in filenames:\n        \n        std_id = filename.split('_')[0]\n        #f_name_id = filename.split('_')[1].split('.')[0]\n        dat_type = dirname.split('/')[-1]\n        img_path = os.path.join(dirname, filename)\n        \n        kf.loc[kf.shape[0]] = [dat_type,std_id,filename,img_path]\n\n\npng_data = pd.DataFrame(kf.drop([0,1,2,3,4],axis=0).values,columns=kf.columns)\npng_data['image_file_ID'] = png_data.image_file_ID.str[13:-4]\npng_data\n\n# creating train and test datframes where all the image_study and studyInstanceUID are stored\npng_test = png_data[png_data['data_type'] == 'test']\npng_train = png_data[png_data['data_type'] == 'train']","metadata":{"execution":{"iopub.status.busy":"2021-06-24T10:22:20.184975Z","iopub.execute_input":"2021-06-24T10:22:20.185394Z","iopub.status.idle":"2021-06-24T10:22:46.143462Z","shell.execute_reply.started":"2021-06-24T10:22:20.185361Z","shell.execute_reply":"2021-06-24T10:22:46.142337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"png_train","metadata":{"execution":{"iopub.status.busy":"2021-06-24T10:22:46.145619Z","iopub.execute_input":"2021-06-24T10:22:46.146085Z","iopub.status.idle":"2021-06-24T10:22:46.169102Z","shell.execute_reply.started":"2021-06-24T10:22:46.146044Z","shell.execute_reply":"2021-06-24T10:22:46.168153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"png_test","metadata":{"execution":{"iopub.status.busy":"2021-06-24T10:22:46.170759Z","iopub.execute_input":"2021-06-24T10:22:46.171412Z","iopub.status.idle":"2021-06-24T10:22:46.189061Z","shell.execute_reply.started":"2021-06-24T10:22:46.171361Z","shell.execute_reply":"2021-06-24T10:22:46.187807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = pd.read_csv(\"/kaggle/input/siim-covid19-detection/sample_submission.csv\")\ntrain_image = pd.read_csv(\"/kaggle/input/siim-covid19-detection/train_image_level.csv\")\ntrain_study = pd.read_csv(\"/kaggle/input/siim-covid19-detection/train_study_level.csv\")\n\n# class label colum and case study id_number columns are created\n\ntrain_study['classification_class'] = np.zeros((train_study.shape[0],1))\nfor i in range(0,train_study.shape[0]):\n    train_study['classification_class'][i] = train_study.iloc[i].values[1:-1]\n\n# StudyInstanceUID \ntrain_study['StudyInstanceUID'] = train_study.id.str[0:-6]\ntrain_study","metadata":{"execution":{"iopub.status.busy":"2021-06-24T10:22:46.191283Z","iopub.execute_input":"2021-06-24T10:22:46.191674Z","iopub.status.idle":"2021-06-24T10:22:50.360823Z","shell.execute_reply.started":"2021-06-24T10:22:46.191632Z","shell.execute_reply":"2021-06-24T10:22:50.359643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"itr = 0\ncl = []\nfor i in train_study['classification_class'].values:\n    if i[0]==1:\n        cl.append('Negative')\n    elif i[1]==1:\n        cl.append('Typical')\n    elif i[2]==1:\n        cl.append('Indeterminate')\n    elif i[3]==1:\n        cl.append('Atypical')\n    else:\n        a=1\n    itr = itr+1\n\ntrain_study['class_type'] = cl\ntrain_study","metadata":{"execution":{"iopub.status.busy":"2021-06-24T10:22:50.362240Z","iopub.execute_input":"2021-06-24T10:22:50.362550Z","iopub.status.idle":"2021-06-24T10:22:50.403277Z","shell.execute_reply.started":"2021-06-24T10:22:50.362518Z","shell.execute_reply":"2021-06-24T10:22:50.402202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Selecting appropriate image for the StudyInstanceUID from coressponding multiple images_id's\n\nSIUID = train_image.groupby('StudyInstanceUID') #SIUID :- Syudy instance UID indexed dataframe\nim_fl_id = png_train.set_index('image_file_ID')\n\nImage_file_path = []\nImage_file_ID = []\nfor i in train_study['StudyInstanceUID'].values:\n    \n    pk = SIUID.get_group(i)\n    if (pk.shape[0] >1) & (pk.dropna(subset=['boxes']).shape[0]>0):\n        \n        im_id = SIUID.get_group(i).dropna(subset=['boxes']).values[0][0].split('_')[0]\n        path = im_fl_id.loc[im_id]['image_file_path']\n        Image_file_path.append(path)\n        Image_file_ID.append(im_id)\n    \n    else:\n        im_id = SIUID.get_group(i).values[0][0].split('_')[0]\n        path = im_fl_id.loc[im_id]['image_file_path']\n        Image_file_path.append(path)\n        Image_file_ID.append(im_id)\n\ntrain_study['image_file_path'] = Image_file_path\ntrain_study['image_file_ID'] = Image_file_ID\ntrain_study","metadata":{"execution":{"iopub.status.busy":"2021-06-24T10:22:50.404657Z","iopub.execute_input":"2021-06-24T10:22:50.405010Z","iopub.status.idle":"2021-06-24T10:23:03.256965Z","shell.execute_reply.started":"2021-06-24T10:22:50.404973Z","shell.execute_reply":"2021-06-24T10:23:03.255911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# using this information for classification problem alone\ntrain_study.info()","metadata":{"execution":{"iopub.status.busy":"2021-06-24T10:23:03.259700Z","iopub.execute_input":"2021-06-24T10:23:03.260437Z","iopub.status.idle":"2021-06-24T10:23:03.280763Z","shell.execute_reply.started":"2021-06-24T10:23:03.260386Z","shell.execute_reply":"2021-06-24T10:23:03.279374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This data frame is for the object localisation and classification combined activities\n\ntrain_Study_image = pd.merge(train_image,train_study,on='StudyInstanceUID',how=\"inner\")\n#\ntrain_Study_image.columns = ['Image_id', 'boxes', 'label', 'StudyInstanceUID', 'Study_id',\n       'Negative for Pneumonia', 'Typical Appearance',\n       'Indeterminate Appearance', 'Atypical Appearance',\n       'classification_class','class_type','image_file_path','image_file_ID']\n\ntrain_Study_image","metadata":{"execution":{"iopub.status.busy":"2021-06-24T10:23:03.282653Z","iopub.execute_input":"2021-06-24T10:23:03.283152Z","iopub.status.idle":"2021-06-24T10:23:03.328498Z","shell.execute_reply.started":"2021-06-24T10:23:03.283101Z","shell.execute_reply":"2021-06-24T10:23:03.327317Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_Study_image.to_csv(r'Modified_train_study_Image_File_Study.csv')\npd.read_csv('Modified_train_study_Image_File_Study.csv')","metadata":{"execution":{"iopub.status.busy":"2021-06-24T10:30:38.465798Z","iopub.execute_input":"2021-06-24T10:30:38.466210Z","iopub.status.idle":"2021-06-24T10:30:38.868488Z","shell.execute_reply.started":"2021-06-24T10:30:38.466167Z","shell.execute_reply":"2021-06-24T10:30:38.867372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Study Instance ID where all the images of the study ID are labelled as none\ntrain_Study_image.set_index('StudyInstanceUID').loc['0d962bb4c9f2']","metadata":{"execution":{"iopub.status.busy":"2021-06-24T10:23:03.329888Z","iopub.execute_input":"2021-06-24T10:23:03.330327Z","iopub.status.idle":"2021-06-24T10:23:03.355609Z","shell.execute_reply.started":"2021-06-24T10:23:03.330288Z","shell.execute_reply":"2021-06-24T10:23:03.354779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = [] \nfor i in train_study['image_file_path'].values:\n    \n    pt = tf.keras.preprocessing.image.load_img(i)\n    kt = tf.keras.preprocessing.image.img_to_array(pt)[40:230,40:230] # image cropping\n    km = tf.keras.preprocessing.image.array_to_img(kt).resize([256,256])\n    image_path_crp = '/kaggle/working' + '/'+ i.split('/')[-1]\n    path.append(image_path_crp)\n    km.save(image_path_crp)\n    \ntrain_study['Key_image_path'] = path","metadata":{"execution":{"iopub.status.busy":"2021-06-24T10:23:03.356631Z","iopub.execute_input":"2021-06-24T10:23:03.357048Z","iopub.status.idle":"2021-06-24T10:25:51.968279Z","shell.execute_reply.started":"2021-06-24T10:23:03.357016Z","shell.execute_reply":"2021-06-24T10:25:51.967166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_study","metadata":{"execution":{"iopub.status.busy":"2021-06-24T10:25:51.970036Z","iopub.execute_input":"2021-06-24T10:25:51.970400Z","iopub.status.idle":"2021-06-24T10:25:52.001425Z","shell.execute_reply.started":"2021-06-24T10:25:51.970365Z","shell.execute_reply":"2021-06-24T10:25:51.999686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_study.to_csv(r'Modified_train_study_Covid.csv')","metadata":{"execution":{"iopub.status.busy":"2021-06-24T10:27:36.105416Z","iopub.execute_input":"2021-06-24T10:27:36.105811Z","iopub.status.idle":"2021-06-24T10:27:36.381351Z","shell.execute_reply.started":"2021-06-24T10:27:36.105778Z","shell.execute_reply":"2021-06-24T10:27:36.380249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('/kaggle/working/Modified_train_study_Covid.csv')","metadata":{"execution":{"iopub.status.busy":"2021-06-24T10:27:40.850209Z","iopub.execute_input":"2021-06-24T10:27:40.850616Z","iopub.status.idle":"2021-06-24T10:27:40.904966Z","shell.execute_reply.started":"2021-06-24T10:27:40.850574Z","shell.execute_reply":"2021-06-24T10:27:40.903895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}