{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Original code from: https://www.kaggle.com/ayuraj/submission-covid19/output\n\n**NOTES**\n1. Re-run of the code is done when submitting. Test folder is modified (hidden elements are added), therefore CAN´T submit csv file directly. \n2. Prediction must be done in Kaggle Notebook.\n3. sample_submission.csv is used to verify the code extracts the correct id's from test folder.\n4. This notebook sets all predicionts to \"negative 1 0 0 1 1\" in the study level and \"none 1 0 0 1 1\" in the case level.\n5. ONLY FOR SUBMISSION TESTING.","metadata":{}},{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-28T13:02:12.313783Z","iopub.execute_input":"2021-06-28T13:02:12.314268Z","iopub.status.idle":"2021-06-28T13:02:12.320153Z","shell.execute_reply.started":"2021-06-28T13:02:12.314229Z","shell.execute_reply":"2021-06-28T13:02:12.319029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Submission ","metadata":{}},{"cell_type":"code","source":"# Read the submisison file\nsub_df = pd.read_csv('/kaggle/input/siim-covid19-detection/sample_submission.csv')\nprint(len(sub_df))\nsub_df","metadata":{"execution":{"iopub.status.busy":"2021-06-28T13:02:12.321797Z","iopub.execute_input":"2021-06-28T13:02:12.322376Z","iopub.status.idle":"2021-06-28T13:02:12.355031Z","shell.execute_reply.started":"2021-06-28T13:02:12.322340Z","shell.execute_reply":"2021-06-28T13:02:12.354292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_df = sub_df.loc[sub_df.id.str.contains('_study')]\nlen(study_df)","metadata":{"execution":{"iopub.status.busy":"2021-06-28T13:02:12.356739Z","iopub.execute_input":"2021-06-28T13:02:12.357081Z","iopub.status.idle":"2021-06-28T13:02:12.366090Z","shell.execute_reply.started":"2021-06-28T13:02:12.357046Z","shell.execute_reply":"2021-06-28T13:02:12.365065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_df = sub_df.loc[sub_df.id.str.contains('_image')]\nlen(image_df)","metadata":{"execution":{"iopub.status.busy":"2021-06-28T13:02:12.367765Z","iopub.execute_input":"2021-06-28T13:02:12.368127Z","iopub.status.idle":"2021-06-28T13:02:12.380421Z","shell.execute_reply.started":"2021-06-28T13:02:12.368092Z","shell.execute_reply":"2021-06-28T13:02:12.379438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Function changed from original\ndef prepare_test_images():\n    image_id = []\n\n    for dirname, _, filenames in tqdm(os.walk(f'../input/siim-covid19-detection/test')):\n        for file in filenames:\n            image_id.append(file.replace('.dcm', ''))\n\n    return image_id","metadata":{"execution":{"iopub.status.busy":"2021-06-28T13:02:12.381568Z","iopub.execute_input":"2021-06-28T13:02:12.381808Z","iopub.status.idle":"2021-06-28T13:02:12.390082Z","shell.execute_reply.started":"2021-06-28T13:02:12.381785Z","shell.execute_reply":"2021-06-28T13:02:12.389255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare Image Level Test Images","metadata":{}},{"cell_type":"code","source":"image_ids = prepare_test_images()\nprint(f'Number of test images: {len(image_ids)}')","metadata":{"execution":{"iopub.status.busy":"2021-06-28T13:02:12.392769Z","iopub.execute_input":"2021-06-28T13:02:12.393011Z","iopub.status.idle":"2021-06-28T13:02:13.956507Z","shell.execute_reply.started":"2021-06-28T13:02:12.392984Z","shell.execute_reply":"2021-06-28T13:02:13.953371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_df = pd.DataFrame.from_dict({'image_id': image_ids})\n\n# Associate image-level id with study-level ids.\n# Note that a study-level might have more than one image-level ids.\nfor study_dir in os.listdir('../input/siim-covid19-detection/test'):\n    for series in os.listdir(f'../input/siim-covid19-detection/test/{study_dir}'):\n        for image in os.listdir(f'../input/siim-covid19-detection/test/{study_dir}/{series}/'):\n            image_id = image[:-4]\n            meta_df.loc[meta_df['image_id'] == image_id, 'study_id'] = study_dir\n\nmeta_df","metadata":{"execution":{"iopub.status.busy":"2021-06-28T13:02:13.961732Z","iopub.execute_input":"2021-06-28T13:02:13.962117Z","iopub.status.idle":"2021-06-28T13:02:15.751993Z","shell.execute_reply.started":"2021-06-28T13:02:13.962079Z","shell.execute_reply":"2021-06-28T13:02:15.751207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_labels = []\n\nfor image_id, df in meta_df.groupby('image_id'):\n    image_labels.append(f'{image_id}_image')\n\nimage_pred = ['none 1 0 0 1 1']*len(image_labels)\nimage_df = pd.DataFrame.from_dict({'id': image_labels, 'PredictionString': image_pred})\nimage_df","metadata":{"execution":{"iopub.status.busy":"2021-06-28T13:02:15.754181Z","iopub.execute_input":"2021-06-28T13:02:15.754549Z","iopub.status.idle":"2021-06-28T13:02:15.794428Z","shell.execute_reply.started":"2021-06-28T13:02:15.754511Z","shell.execute_reply":"2021-06-28T13:02:15.793691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_labels = []\n\nfor study_id, df in meta_df.groupby('study_id'):\n    study_labels.append(f'{study_id}_study')\n\nstudy_pred = ['negative 1 0 0 1 1']*len(study_labels)\nstudy_df = pd.DataFrame.from_dict({'id': study_labels, 'PredictionString': study_pred})\nstudy_df","metadata":{"execution":{"iopub.status.busy":"2021-06-28T13:02:15.796998Z","iopub.execute_input":"2021-06-28T13:02:15.797263Z","iopub.status.idle":"2021-06-28T13:02:15.834871Z","shell.execute_reply.started":"2021-06-28T13:02:15.797238Z","shell.execute_reply":"2021-06-28T13:02:15.834163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = pd.concat([study_df, image_df])\nsub_df.to_csv('submission.csv', index=False)\nsub_df","metadata":{"execution":{"iopub.status.busy":"2021-06-28T13:02:15.836008Z","iopub.execute_input":"2021-06-28T13:02:15.836353Z","iopub.status.idle":"2021-06-28T13:02:15.856987Z","shell.execute_reply.started":"2021-06-28T13:02:15.836319Z","shell.execute_reply":"2021-06-28T13:02:15.856053Z"},"trusted":true},"execution_count":null,"outputs":[]}]}