{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport random \nfrom shutil import copyfile\nimport pydicom as dicom\nimport cv2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#from pathlib import Path\n\n#train = Path(\"data/train/\")\n#test = Path(\"data/test/\")\n#train.mkdir(parents=True, exist_ok=True)\n#test.mkdir(parents=True, exist_ok=True)\n#os.mkdir('/kaggle/working/rsna_converted_stage_2_train_images/')\n#os.mkdir('/kaggle/working/rsna_converted_stage_2_test_images/')\nos.mkdir('/kaggle/working/train')\n#os.mkdir('dataset/train')\nos.mkdir('/kaggle/working/test')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# set parameters here\nsavepath = '/kaggle/working'\nseed = 0\nnp.random.seed(seed) # Reset the seed so all runs are the same.\nrandom.seed(seed)\nMAXVAL = 255  # Range [0 255]\n\n# path to covid-19 dataset from https://github.com/ieee8023/covid-chestxray-dataset\ncohen_imgpath = '/kaggle/input/covidchestxraydataset/images' \ncohen_csvpath = '/kaggle/input/covidchestxraydataset/metadata.csv'\n\n# path to covid-19 dataset from https://github.com/agchung/Figure1-COVID-chestxray-dataset\nfig1_imgpath = '/kaggle/input/figure1covidchestxraydataset/images'\nfig1_csvpath = '/kaggle/input/figure1covidchestxraydataset/metadata.csv'\n\n# path to covid-19 dataset from https://github.com/agchung/Actualmed-COVID-chestxray-dataset\nactmed_imgpath = '/kaggle/input/actualmedcovidchestxraydataset/images'\nactmed_csvpath = '/kaggle/input/actualmedcovidchestxraydataset/metadata.csv'\n\n# path to covid-19 dataset from https://www.kaggle.com/tawsifurrahman/covid19-radiography-database\nsirm_imgpath = '/kaggle/input/covid19-radiography-database/COVID-19 Radiography Database/COVID-19'\nsirm_csvpath = '/kaggle/input/covid19-radiography-database/COVID-19 Radiography Database/COVID-19.metadata.xlsx'\n\n# path to https://www.kaggle.com/c/rsna-pneumonia-detection-challenge\n#rsna_datapath = '/kaggle/input/rsna-pneumonia-detection-challenge'\nrsna_datapath = '/kaggle/input/rsna-pneumonia-detection-challenge/'\n# get all the normal from here\nrsna_csvname = 'stage_2_detailed_class_info.csv' \n# get all the 1s from here since 1 indicate pneumonia\n# found that images that aren't pneunmonia and also not normal are classified as 0s\nrsna_csvname2 = 'stage_2_train_labels.csv' \nrsna_imgpath = 'stage_2_train_images'\n\n# parameters for COVIDx dataset\ntrain = []\ntest = []\ntest_count = {'normal': 0, 'pneumonia': 0, 'COVID-19': 0}\ntrain_count = {'normal': 0, 'pneumonia': 0, 'COVID-19': 0}\n\nmapping = dict()\nmapping['COVID-19'] = 'COVID-19'\nmapping['SARS'] = 'pneumonia'\nmapping['MERS'] = 'pneumonia'\nmapping['Streptococcus'] = 'pneumonia'\nmapping['Klebsiella'] = 'pneumonia'\nmapping['Chlamydophila'] = 'pneumonia'\nmapping['Legionella'] = 'pneumonia'\nmapping['Normal'] = 'normal'\nmapping['Lung Opacity'] = 'pneumonia'\nmapping['1'] = 'pneumonia'\n\n# train/test split\nsplit = 0.1\n\n# to avoid duplicates\npatient_imgpath = {}\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# adapted from https://github.com/mlmed/torchxrayvision/blob/master/torchxrayvision/datasets.py#L814\ncohen_csv = pd.read_csv(cohen_csvpath, nrows=None)\n#idx_pa = csv[\"view\"] == \"PA\"  # Keep only the PA view\nviews = [\"PA\", \"AP\", \"AP Supine\", \"AP semi erect\", \"AP erect\"]\ncohen_idx_keep = cohen_csv.view.isin(views)\ncohen_csv = cohen_csv[cohen_idx_keep]\n\nfig1_csv = pd.read_csv(fig1_csvpath, encoding='ISO-8859-1', nrows=None)\nactmed_csv = pd.read_csv(actmed_csvpath, nrows=None)\n\nsirm_csv = pd.read_excel(sirm_csvpath)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# get non-COVID19 viral, bacteria, and COVID-19 infections from covid-chestxray-dataset, figure1 and actualmed\n# stored as patient id, image filename and label\nfilename_label = {'normal': [], 'pneumonia': [], 'COVID-19': []}\ncount = {'normal': 0, 'pneumonia': 0, 'COVID-19': 0}\ncovid_ds = {'cohen': [], 'fig1': [], 'actmed': [], 'sirm': []}\n\nfor index, row in cohen_csv.iterrows():\n    f = row['finding'].split(',')[0] # take the first finding, for the case of COVID-19, ARDS\n    if f in mapping: # \n        count[mapping[f]] += 1\n        entry = [str(row['patientid']), row['filename'], mapping[f], 'cohen']\n        filename_label[mapping[f]].append(entry)\n        if mapping[f] == 'COVID-19':\n            covid_ds['cohen'].append(str(row['patientid']))\n        \nfor index, row in fig1_csv.iterrows():\n    if not str(row['finding']) == 'nan':\n        f = row['finding'].split(',')[0] # take the first finding\n        if f in mapping: # \n            count[mapping[f]] += 1\n            if os.path.exists(os.path.join(fig1_imgpath, row['patientid'] + '.jpg')):\n                entry = [row['patientid'], row['patientid'] + '.jpg', mapping[f], 'fig1']\n            elif os.path.exists(os.path.join(fig1_imgpath, row['patientid'] + '.png')):\n                entry = [row['patientid'], row['patientid'] + '.png', mapping[f], 'fig1']\n            filename_label[mapping[f]].append(entry)\n            if mapping[f] == 'COVID-19':\n                covid_ds['fig1'].append(row['patientid'])\n\nfor index, row in actmed_csv.iterrows():\n    if not str(row['finding']) == 'nan':\n        f = row['finding'].split(',')[0]\n        if f in mapping:\n            count[mapping[f]] += 1\n            entry = [row['patientid'], row['imagename'], mapping[f], 'actmed']\n            filename_label[mapping[f]].append(entry)\n            if mapping[f] == 'COVID-19':\n                covid_ds['actmed'].append(row['patientid'])\n    \nsirm = set(sirm_csv['URL'])\ncohen = set(cohen_csv['url'])\ndiscard = ['100', '101', '102', '103', '104', '105', \n           '110', '111', '112', '113', '122', '123', \n           '124', '125', '126', '217']\n\nfor idx, row in sirm_csv.iterrows():\n    patientid = row['FILE NAME']\n    if row['URL'] not in cohen and patientid[patientid.find('(')+1:patientid.find(')')] not in discard:\n        count[mapping['COVID-19']] += 1\n        imagename = patientid + '.' + row['FORMAT'].lower()\n        if not os.path.exists(os.path.join(sirm_imgpath, imagename)):\n            imagename = patientid.split('(')[0] + ' ('+ patientid.split('(')[1] + '.' + row['FORMAT'].lower()\n        entry = [patientid, imagename, mapping['COVID-19'], 'sirm']\n        filename_label[mapping['COVID-19']].append(entry)\n        covid_ds['sirm'].append(patientid)\n    \nprint('Data distribution from covid datasets:')\nprint(count)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# add covid-chestxray-dataset, figure1 and actualmed into COVIDx dataset\n# since these datasets don't have test dataset, split into train/test by patientid\n# for covid-chestxray-dataset:\n# patient 8 is used as non-COVID19 viral test\n# patient 31 is used as bacterial test\n# patients 19, 20, 36, 42, 86 are used as COVID-19 viral test\n# for figure 1:\n# patients 24, 25, 27, 29, 30, 32, 33, 36, 37, 38\n\nds_imgpath = {'cohen': cohen_imgpath, 'fig1': fig1_imgpath, 'actmed': actmed_imgpath, 'sirm': sirm_imgpath}\n\nfor key in filename_label.keys():\n    arr = np.array(filename_label[key])\n    if arr.size == 0:\n        continue\n    # split by patients\n    # num_diff_patients = len(np.unique(arr[:,0]))\n    # num_test = max(1, round(split*num_diff_patients))\n    # select num_test number of random patients\n    # random.sample(list(arr[:,0]), num_test)\n    if key == 'pneumonia':\n        test_patients = ['8', '31']\n    elif key == 'COVID-19':\n        test_patients = ['19', '20', '36', '42', '86', \n                         '94', '97', '117', '132', \n                         '138', '144', '150', '163', '169', '174', '175', '179', '190', '191'\n                         'COVID-00024', 'COVID-00025', 'COVID-00026', 'COVID-00027', 'COVID-00029',\n                         'COVID-00030', 'COVID-00032', 'COVID-00033', 'COVID-00035', 'COVID-00036',\n                         'COVID-00037', 'COVID-00038',\n                         'ANON24', 'ANON45', 'ANON126', 'ANON106', 'ANON67',\n                         'ANON153', 'ANON135', 'ANON44', 'ANON29', 'ANON201', \n                         'ANON191', 'ANON234', 'ANON110', 'ANON112', 'ANON73', \n                         'ANON220', 'ANON189', 'ANON30', 'ANON53', 'ANON46',\n                         'ANON218', 'ANON240', 'ANON100', 'ANON237', 'ANON158',\n                         'ANON174', 'ANON19', 'ANON195',\n                         'COVID-19(119)', 'COVID-19(87)', 'COVID-19(70)', 'COVID-19(94)', \n                         'COVID-19(215)', 'COVID-19(77)', 'COVID-19(213)', 'COVID-19(81)', \n                         'COVID-19(216)', 'COVID-19(72)', 'COVID-19(106)', 'COVID-19(131)', \n                         'COVID-19(107)', 'COVID-19(116)', 'COVID-19(95)', 'COVID-19(214)', \n                         'COVID-19(129)']\n    else: \n        test_patients = []\n    print('Key: ', key)\n    print('Test patients: ', test_patients)\n    # go through all the patients\n    for patient in arr:\n        if patient[0] not in patient_imgpath:\n            patient_imgpath[patient[0]] = [patient[1]]\n        else:\n            if patient[1] not in patient_imgpath[patient[0]]:\n                patient_imgpath[patient[0]].append(patient[1])\n            else:\n                continue  # skip since image has already been written\n        if patient[0] in test_patients:\n            if patient[3] == 'sirm':\n                image = cv2.imread(os.path.join(ds_imgpath[patient[3]], patient[1]))\n                gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)\n                patient[1] = patient[1].replace(' ', '')\n                cv2.imwrite(os.path.join(savepath, 'test', patient[1]), gray)\n            else:\n                copyfile(os.path.join(ds_imgpath[patient[3]], patient[1]), os.path.join(savepath, 'test', patient[1]))\n            test.append(patient)\n            test_count[patient[2]] += 1\n        else:\n            if patient[3] == 'sirm':\n                image = cv2.imread(os.path.join(ds_imgpath[patient[3]], patient[1]))\n                gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)\n                patient[1] = patient[1].replace(' ', '')\n                cv2.imwrite(os.path.join(savepath, 'train', patient[1]), gray)\n            else:\n                copyfile(os.path.join(ds_imgpath[patient[3]], patient[1]), os.path.join(savepath, 'train', patient[1]))\n            train.append(patient)\n            train_count[patient[2]] += 1\n\nprint('test count: ', test_count)\nprint('train count: ', train_count)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#add normal and rest of pneumonia cases from https://www.kaggle.com/c/rsna-pneumonia-detection-challenge\ncsv_normal = pd.read_csv(os.path.join(rsna_datapath, rsna_csvname), nrows=None)\ncsv_pneu = pd.read_csv(os.path.join(rsna_datapath, rsna_csvname2), nrows=None)\npatients = {'normal': [], 'pneumonia': []}\n\nfor index, row in csv_normal.iterrows():\n    if row['class'] == 'Normal':\n        patients['normal'].append(row['patientId'])\n\nfor index, row in csv_pneu.iterrows():\n    if int(row['Target']) == 1:\n        patients['pneumonia'].append(row['patientId'])\n\nfor key in patients.keys():\n    arr = np.array(patients[key])\n    if arr.size == 0:\n        continue\n    # split by patients \n    # num_diff_patients = len(np.unique(arr))\n    # num_test = max(1, round(split*num_diff_patients))\n    test_patients = np.load('rsna_test_patients_{}.npy'.format(key)) # random.sample(list(arr), num_test), download the .npy files from the repo.\n    # np.save('rsna_test_patients_{}.npy'.format(key), np.array(test_patients))\n    for patient in arr:\n        if patient not in patient_imgpath:\n            patient_imgpath[patient] = [patient]\n        else:\n            continue  # skip since image has already been written\n                \n        ds = dicom.dcmread(os.path.join(rsna_datapath, rsna_imgpath, patient + '.dcm'))\n        pixel_array_numpy = ds.pixel_array\n        imgname = patient + '.png'\n        if patient in test_patients:\n            cv2.imwrite(os.path.join(savepath, 'test', imgname), pixel_array_numpy)\n            test.append([patient, imgname, key, 'rsna'])\n            test_count[key] += 1\n        else:\n            cv2.imwrite(os.path.join(savepath, 'train', imgname), pixel_array_numpy)\n            train.append([patient, imgname, key, 'rsna'])\n            train_count[key] += 1\n\nprint('test count: ', test_count)\nprint('train count: ', train_count)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"arr","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# final stats\nprint('Final stats')\nprint('Train count: ', train_count)\nprint('Test count: ', test_count)\nprint('Total length of train: ', len(train))\nprint('Total length of test: ', len(test))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# export to train and test csv\n# format as patientid, filename, label, separated by a space\ntrain_file = open(\"train_split_v3.txt\",'w') \nfor sample in train:\n    if len(sample) == 4:\n        info = str(sample[0]) + ' ' + sample[1] + ' ' + sample[2] + ' ' + sample[3] + '\\n'\n    else:\n        info = str(sample[0]) + ' ' + sample[1] + ' ' + sample[2] + '\\n'\n    train_file.write(info)\n\ntrain_file.close()\n\ntest_file = open(\"test_split_v3.txt\", 'w')\nfor sample in test:\n    if len(sample) == 4:\n        info = str(sample[0]) + ' ' + sample[1] + ' ' + sample[2] + ' ' + sample[3] + '\\n'\n    else:\n        info = str(sample[0]) + ' ' + sample[1] + ' ' + sample[2] + '\\n'\n    test_file.write(info)\n\ntest_file.close()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python (covid)","language":"python","name":"covid"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.10"}},"nbformat":4,"nbformat_minor":4}