{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Thanks to https://www.kaggle.com/xhlulu/ranzcr-efficientnet-tpu-training**\n\n**Let's use the extra data provided by https://www.kaggle.com/raddar/ricord-covid19-xray-positive-tests**\n\n**Thanks to https://www.kaggle.com/josephamigo/siim-external-data-pipeline**","metadata":{}},{"cell_type":"markdown","source":"**I think besides this adding BIMCV to dataset is the key for winning this competition. **","metadata":{}},{"cell_type":"code","source":"!pip install efficientnet -q\n!conda install gdcm -c conda-forge -y","metadata":{"execution":{"iopub.status.busy":"2021-07-08T18:09:05.481868Z","iopub.execute_input":"2021-07-08T18:09:05.482251Z","iopub.status.idle":"2021-07-08T18:09:47.86601Z","shell.execute_reply.started":"2021-07-08T18:09:05.482219Z","shell.execute_reply":"2021-07-08T18:09:47.864894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\nimport efficientnet.tfkeras as efn\nimport numpy as np\nimport pandas as pd\nfrom kaggle_datasets import KaggleDatasets\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom sklearn.model_selection import GroupKFold","metadata":{"execution":{"iopub.status.busy":"2021-07-08T18:09:47.869918Z","iopub.execute_input":"2021-07-08T18:09:47.870244Z","iopub.status.idle":"2021-07-08T18:09:47.876489Z","shell.execute_reply.started":"2021-07-08T18:09:47.870211Z","shell.execute_reply":"2021-07-08T18:09:47.875027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"COMPETITION_NAME = \"siimcovid19-512-img-png-600-study-png\"\nGCS_DS_PATH = KaggleDatasets().get_gcs_path(COMPETITION_NAME)","metadata":{"execution":{"iopub.status.busy":"2021-07-08T18:09:47.878783Z","iopub.execute_input":"2021-07-08T18:09:47.879196Z","iopub.status.idle":"2021-07-08T18:09:48.166991Z","shell.execute_reply.started":"2021-07-08T18:09:47.879156Z","shell.execute_reply":"2021-07-08T18:09:48.165909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"load_dir = f\"/kaggle/input/{COMPETITION_NAME}/\"\ndf = pd.read_csv('../input/siim-covid19-detection/train_study_level.csv')\nlabel_cols = df.columns[1:5]\ndf #Just includes id and numbering of 4 classes","metadata":{"execution":{"iopub.status.busy":"2021-07-08T18:09:48.168924Z","iopub.execute_input":"2021-07-08T18:09:48.169351Z","iopub.status.idle":"2021-07-08T18:09:48.226703Z","shell.execute_reply.started":"2021-07-08T18:09:48.169309Z","shell.execute_reply":"2021-07-08T18:09:48.225926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IM_PATH = '../input/ricord-covid19-xray-positive-tests/MIDRC-RICORD/MIDRC-RICORD'\nmeta_extradata = pd.read_csv('../input/ricord-covid19-xray-positive-tests/MIDRC-RICORD-meta.csv')\nmeta_extradata.dropna(inplace = True, subset = ['labels'])","metadata":{"execution":{"iopub.status.busy":"2021-07-08T18:09:48.227919Z","iopub.execute_input":"2021-07-08T18:09:48.228392Z","iopub.status.idle":"2021-07-08T18:09:48.273917Z","shell.execute_reply.started":"2021-07-08T18:09:48.228361Z","shell.execute_reply":"2021-07-08T18:09:48.272961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_extradata","metadata":{"execution":{"iopub.status.busy":"2021-07-08T18:09:48.275168Z","iopub.execute_input":"2021-07-08T18:09:48.275428Z","iopub.status.idle":"2021-07-08T18:09:48.300254Z","shell.execute_reply.started":"2021-07-08T18:09:48.275402Z","shell.execute_reply":"2021-07-08T18:09:48.299228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def locate_row_to_delete(a):\n    if a.max() <= 0.5 :\n        return np.array([np.nan,np.nan,np.nan,np.nan])\n    else:\n        return a\n\n\ndef encode_labels(df):\n    df = df[['fname', 'labels']]\n    \n    #initialize label columns\n    df['Negative for Pneumonia'] = 0\n    df['Typical Appearance'] = 0\n    df['Indeterminate Appearance'] = 0\n    df['Atypical Appearance'] = 0\n    \n    #Count occurences of each category\n    df['Negative for Pneumonia'] = df.labels.apply(lambda x : x.count('Negative'))\n    df['Typical Appearance'] = df.labels.apply(lambda x : x.count('Typical'))\n    df['Indeterminate Appearance'] = df.labels.apply(lambda x : x.count('Indeterminate'))\n    df['Atypical Appearance'] = df.labels.apply(lambda x : x.count('Atypical'))\n    \n    #df to array for computations\n    labels_np = df[['Negative for Pneumonia','Typical Appearance', 'Indeterminate Appearance', 'Atypical Appearance']].values\n    temp = labels_np/labels_np.sum(axis = 1, keepdims=True)\n    temp = np.apply_along_axis(locate_row_to_delete, 1, temp) \n\n    temp -= 0.51\n    \n    temp[temp > 0] = 1\n    temp[temp < 0] = 0\n    \n\n    df['Negative for Pneumonia'] = temp[:,0]\n    df['Typical Appearance'] = temp[:,1]\n    df['Indeterminate Appearance'] = temp[:,2]\n    df['Atypical Appearance'] = temp[:,3]\n    \n    \n    df.dropna(subset = ['Negative for Pneumonia'], inplace = True)\n    return df","metadata":{"execution":{"iopub.status.busy":"2021-07-08T18:09:48.301628Z","iopub.execute_input":"2021-07-08T18:09:48.301932Z","iopub.status.idle":"2021-07-08T18:09:48.312993Z","shell.execute_reply.started":"2021-07-08T18:09:48.301904Z","shell.execute_reply":"2021-07-08T18:09:48.312227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = encode_labels(meta_extradata)\noutput","metadata":{"execution":{"iopub.status.busy":"2021-07-08T18:09:48.314971Z","iopub.execute_input":"2021-07-08T18:09:48.315392Z","iopub.status.idle":"2021-07-08T18:09:48.376028Z","shell.execute_reply.started":"2021-07-08T18:09:48.315357Z","shell.execute_reply":"2021-07-08T18:09:48.374958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Data still have extra columns and different names**\n* First change the column name from fname -> id\n\n* Then drop NaN \n\n* Then drop labels column","metadata":{}},{"cell_type":"code","source":"output = output.rename(columns={'fname': 'id'})\nnan_value = float(\"NaN\") \noutput. replace(\"\", nan_value, inplace=True)\noutput. dropna(subset = [\"labels\"], inplace=True)\noutput=output.drop(['labels'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-07-08T18:09:48.37805Z","iopub.execute_input":"2021-07-08T18:09:48.378494Z","iopub.status.idle":"2021-07-08T18:09:48.390361Z","shell.execute_reply.started":"2021-07-08T18:09:48.378437Z","shell.execute_reply":"2021-07-08T18:09:48.38963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output=output.reset_index(drop=True)\nprint(output)","metadata":{"execution":{"iopub.status.busy":"2021-07-08T18:09:48.391673Z","iopub.execute_input":"2021-07-08T18:09:48.391996Z","iopub.status.idle":"2021-07-08T18:09:48.411843Z","shell.execute_reply.started":"2021-07-08T18:09:48.391956Z","shell.execute_reply":"2021-07-08T18:09:48.410501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Concatenate the extra files with original dataframe to obatin the extended one**","metadata":{}},{"cell_type":"code","source":"temp_df=[df, output]\nextended = pd.concat(temp_df)\nextended","metadata":{"execution":{"iopub.status.busy":"2021-07-08T18:09:48.413465Z","iopub.execute_input":"2021-07-08T18:09:48.413939Z","iopub.status.idle":"2021-07-08T18:09:48.440781Z","shell.execute_reply.started":"2021-07-08T18:09:48.413893Z","shell.execute_reply":"2021-07-08T18:09:48.439669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"extended=extended.reset_index(drop=True)\nprint(extended)","metadata":{"execution":{"iopub.status.busy":"2021-07-08T18:09:48.441978Z","iopub.execute_input":"2021-07-08T18:09:48.442256Z","iopub.status.idle":"2021-07-08T18:09:48.455094Z","shell.execute_reply.started":"2021-07-08T18:09:48.442222Z","shell.execute_reply":"2021-07-08T18:09:48.453891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Now time to use groupkfold to extended data**","metadata":{}},{"cell_type":"code","source":"gkf  = GroupKFold(n_splits = 5)\nextended['fold'] = -1\nfor fold, (train_idx, val_idx) in enumerate(gkf.split(extended, groups = extended.id.tolist())):\n    extended.loc[val_idx, 'fold'] = fold\nextended\ndf=extended\nprint(df)","metadata":{"execution":{"iopub.status.busy":"2021-07-08T18:09:48.456573Z","iopub.execute_input":"2021-07-08T18:09:48.456903Z","iopub.status.idle":"2021-07-08T18:09:48.531513Z","shell.execute_reply.started":"2021-07-08T18:09:48.456872Z","shell.execute_reply":"2021-07-08T18:09:48.530477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save all the extended data to temporary directory","metadata":{}},{"cell_type":"code","source":"from PIL import Image\nimport os\n\nsave_dir = f'/kaggle/tmp/external/'\nos.makedirs(save_dir, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2021-07-08T18:09:48.534523Z","iopub.execute_input":"2021-07-08T18:09:48.534914Z","iopub.status.idle":"2021-07-08T18:09:48.539869Z","shell.execute_reply.started":"2021-07-08T18:09:48.53488Z","shell.execute_reply":"2021-07-08T18:09:48.538857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tqdm\nfrom PIL import Image\nfrom tqdm.auto import tqdm\n\nfor dirname, _, filenames in tqdm(os.walk(f'../input/siimcovid19-512-img-png-600-study-png/study')):\n    for file in filenames:\n        photo = Image.open(os.path.join(dirname, file))\n        photo.save(os.path.join(save_dir, file.replace('.png', '.jpg')))","metadata":{"execution":{"iopub.status.busy":"2021-07-08T18:09:48.540964Z","iopub.execute_input":"2021-07-08T18:09:48.541255Z","iopub.status.idle":"2021-07-08T18:11:29.644371Z","shell.execute_reply.started":"2021-07-08T18:09:48.541226Z","shell.execute_reply":"2021-07-08T18:11:29.643594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dirname, _, filenames in tqdm(os.walk(f'../input/ricord-covid19-xray-positive-tests/MIDRC-RICORD/MIDRC-RICORD')):\n    for file in filenames:\n        photo = Image.open(os.path.join(dirname, file))\n        newsize = (600, 600) #Make image size 600x600\n        photo = photo.resize(newsize)\n        photo.save(os.path.join(save_dir, file.replace('.dcm.jpg', '.jpg')))","metadata":{"execution":{"iopub.status.busy":"2021-07-08T18:22:00.648613Z","iopub.execute_input":"2021-07-08T18:22:00.649039Z","iopub.status.idle":"2021-07-08T18:24:54.569888Z","shell.execute_reply.started":"2021-07-08T18:22:00.649003Z","shell.execute_reply":"2021-07-08T18:24:54.569023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**If you want to check files uncomment the following**","metadata":{}},{"cell_type":"code","source":"# files = os.listdir(save_dir)\n# files","metadata":{"execution":{"iopub.status.busy":"2021-07-08T18:17:42.89702Z","iopub.execute_input":"2021-07-08T18:17:42.897533Z","iopub.status.idle":"2021-07-08T18:17:42.906383Z","shell.execute_reply.started":"2021-07-08T18:17:42.897478Z","shell.execute_reply":"2021-07-08T18:17:42.905312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv(f'df.csv')\n!tar -zcf test.tar.gz -C \"/kaggle/tmp/external/\" .","metadata":{"execution":{"iopub.status.busy":"2021-07-08T19:10:20.025016Z","iopub.execute_input":"2021-07-08T19:10:20.025413Z","iopub.status.idle":"2021-07-08T19:10:20.101799Z","shell.execute_reply.started":"2021-07-08T19:10:20.025312Z","shell.execute_reply":"2021-07-08T19:10:20.099778Z"},"trusted":true},"execution_count":null,"outputs":[]}]}