{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\n\nimport os\n\nimport cv2 as cv\nimport pydicom\n\nfrom glob import glob\nfrom os.path import join\nfrom tqdm.autonotebook import tqdm\n\n\n# TRAINING SET\n# The training data is provided as a set of patientIds and bounding boxes (x-min y-min width height)\n# There is also a binary target column, Target, indicating pneumonia or non-pneumonia.\n# There may be multiple rows per patientId\n# stage_2_train_labels.csv -> 1st column = patientID, 6th column = 1 if pneumonia / 0 if not\n\n# imgList: to concatenate images into 'stage_2_train_images' in order to obtain a list; each image is named as follows: 'patientId'.dcm\n# infoAboutData: file csv containing information about available data (1st column : patientId, 6th column: 1/0)\n\ndef MakeDataFrame(imgList,infoAboutData):\n    \n    filepath=[]\n    filename=[]\n    label=[]\n    classe=[]\n    \n    # firstly imgList is ordered, then it is used as input\n    # infoAboutData is already ordered, but copied elements are removed before using it as input:\n    index = 0 # patients' indexes coincide\n    for imagePath in imgList:\n        \n        #path = tf.strings.split(imagePath)\n        filepath.append(imagePath)\n        \n        image = (tf.strings.split(imagePath, os.path.sep))[-1]\n        filename.append(image)\n        target = infoAboutData.iloc[index]['Target']\n        if target == 1:\n            label.append(1)\n            classe.append('PNEUMONIA')\n        elif target==0:\n            label.append(0)\n            classe.append('NORMAL')\n        \n        index+=1\n        \n    dataFrame=pd.DataFrame({\n        \"path\":filepath,\n        \"X\":filename,\n        \"y\":label,\n        \"class\":classe\n        })\n    \n    return dataFrame","metadata":{"execution":{"iopub.status.busy":"2022-01-07T14:28:13.525716Z","iopub.execute_input":"2022-01-07T14:28:13.526771Z","iopub.status.idle":"2022-01-07T14:28:17.926989Z","shell.execute_reply.started":"2022-01-07T14:28:13.526655Z","shell.execute_reply":"2022-01-07T14:28:17.926242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUM_IMG=2000\nDOWNLOAD_TEST = True\n","metadata":{"execution":{"iopub.status.busy":"2022-01-07T14:28:17.92874Z","iopub.execute_input":"2022-01-07T14:28:17.929461Z","iopub.status.idle":"2022-01-07T14:28:17.933226Z","shell.execute_reply.started":"2022-01-07T14:28:17.929422Z","shell.execute_reply":"2022-01-07T14:28:17.932406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make a list with all the names of the images of the training set\nimgNames = sorted(glob('../input/rsna-pneumonia-detection-challenge/stage_2_train_images/*')) \ninfo = pd.read_csv('../input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv')\ninfo = info.drop_duplicates(subset=['patientId'],ignore_index=True)\ninfo=info.sort_values(by=['patientId'])\n\ndf=MakeDataFrame(imgNames,info)","metadata":{"execution":{"iopub.status.busy":"2022-01-07T14:28:17.934574Z","iopub.execute_input":"2022-01-07T14:28:17.934842Z","iopub.status.idle":"2022-01-07T14:30:26.861992Z","shell.execute_reply.started":"2022-01-07T14:28:17.934803Z","shell.execute_reply":"2022-01-07T14:30:26.861219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df.shape)\ndf.head(30)","metadata":{"execution":{"iopub.status.busy":"2022-01-07T14:30:26.863306Z","iopub.execute_input":"2022-01-07T14:30:26.863565Z","iopub.status.idle":"2022-01-07T14:30:26.885015Z","shell.execute_reply.started":"2022-01-07T14:30:26.863532Z","shell.execute_reply":"2022-01-07T14:30:26.88424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SAMPLING: to choose randomly NUM_IMG pneumonia and NUM_IMG healthy samples.\n# (axis = 0 since raws, and so patients, are selected)\n\ntot_pneumonia = df.query('y==1')\nsamp_pneumonia = tot_pneumonia.sample(n=NUM_IMG, axis=0)\nprint(samp_pneumonia.shape)\nprint(samp_pneumonia.head())\nprint()\n\ntot_normal = df.query('y==0')\nsamp_normal = tot_normal.sample(n=NUM_IMG, axis=0)\nprint(samp_normal.shape)\nprint(samp_normal.head())","metadata":{"execution":{"iopub.status.busy":"2022-01-07T14:30:26.887029Z","iopub.execute_input":"2022-01-07T14:30:26.887413Z","iopub.status.idle":"2022-01-07T14:30:26.912454Z","shell.execute_reply.started":"2022-01-07T14:30:26.887376Z","shell.execute_reply":"2022-01-07T14:30:26.911812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#pydicom.dcmread('path/to/file')\ndef load_image(fname):\n    img_dcm = pydicom.dcmread(fname)\n    img_np = img_dcm.pixel_array\n    # img_np is a numpy array\n    return img_np","metadata":{"execution":{"iopub.status.busy":"2022-01-07T14:30:26.91357Z","iopub.execute_input":"2022-01-07T14:30:26.913802Z","iopub.status.idle":"2022-01-07T14:30:26.920381Z","shell.execute_reply.started":"2022-01-07T14:30:26.913771Z","shell.execute_reply":"2022-01-07T14:30:26.9196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def scarico_dati(fname, y, n, folder):\n    img = load_image(fname)\n    img_name = fname.split(\"/\")[-1].split(\".\")[0]\n    path_NORM=join(folder,'Normale')\n    path_PNEU=join(folder,'Pneumonia')\n    if not os.path.exists(path_NORM):\n        os.makedirs(path_NORM)\n    if not os.path.exists(path_PNEU):\n        os.makedirs(path_PNEU)\n    # Pneumonia\n    if y==1:\n        out_file_pneu = join(path_PNEU, f\"{img_name}_label{y}.jpg\")\n        cv.imwrite(out_file_pneu, img)\n    # Normal\n    elif y==0:\n        out_file_norm = join(path_NORM, f\"{img_name}_label{y}.jpg\")\n        cv.imwrite(out_file_norm, img)\n        \n    return","metadata":{"execution":{"iopub.status.busy":"2022-01-07T14:30:26.92181Z","iopub.execute_input":"2022-01-07T14:30:26.922094Z","iopub.status.idle":"2022-01-07T14:30:26.929765Z","shell.execute_reply.started":"2022-01-07T14:30:26.922057Z","shell.execute_reply":"2022-01-07T14:30:26.928935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_test=\"/kaggle/working/Test\"\nif not os.path.exists(path_test):   \n    os.makedirs(path_test)\nelse:\n    print('the file exist')\nif DOWNLOAD_TEST:\n        for fname,y in tqdm(samp_pneumonia[[\"path\",\"y\"]].values):\n            scarico_dati(fname, y, n=1, folder=path_test) \n        for fname,y in tqdm(samp_normal[[\"path\",\"y\"]].values):\n            scarico_dati(fname, y, n=1, folder=path_test) ","metadata":{"execution":{"iopub.status.busy":"2022-01-07T14:30:26.931499Z","iopub.execute_input":"2022-01-07T14:30:26.931907Z","iopub.status.idle":"2022-01-07T14:32:07.835744Z","shell.execute_reply.started":"2022-01-07T14:30:26.931781Z","shell.execute_reply":"2022-01-07T14:32:07.835094Z"},"trusted":true},"execution_count":null,"outputs":[]}]}