{"cells":[{"metadata":{"_cell_guid":"6ac85b1b-2d18-452a-a5df-98404e0afcad","_uuid":"7b54cbda-281a-471f-9b79-fd1405a54ab4","execution":{"iopub.execute_input":"2020-09-12T09:03:54.610842Z","iopub.status.busy":"2020-09-12T09:03:54.610045Z","iopub.status.idle":"2020-09-12T09:05:07.218610Z","shell.execute_reply":"2020-09-12T09:05:07.217881Z"},"papermill":{"duration":72.622674,"end_time":"2020-09-12T09:05:07.218783","exception":false,"start_time":"2020-09-12T09:03:54.596109","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"!conda install -c conda-forge gdcm -y","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"afe1d81e-0cdd-4b3e-a436-d23143b144fb","_uuid":"5d1cae34-a9f9-43f7-96c8-d843f70cd282","execution":{"iopub.execute_input":"2020-09-12T09:05:07.404218Z","iopub.status.busy":"2020-09-12T09:05:07.403485Z","iopub.status.idle":"2020-09-12T09:05:16.793138Z","shell.execute_reply":"2020-09-12T09:05:16.794026Z"},"papermill":{"duration":9.483646,"end_time":"2020-09-12T09:05:16.794325","exception":false,"start_time":"2020-09-12T09:05:07.310679","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"#This script will be used create masks on the images prior to further processing. This will speed up the process\n#of training.\n\nimport ct_scan_processing as ct\nimport osic_utils\nimport numpy as np\nimport tensorflow as tf\nimport pandas as pd\nimport os\nimport pickle\nimport pydicom\nimport gzip\nfrom time import time as time\n\ndir_img = 'train-image-processed'\nif dir_img not in os.listdir():\n    os.mkdir(dir_img)\n\n# obtain the ids associated with the training set\ntrain_data = pd.read_csv(\"/kaggle/input/osic-pulmonary-fibrosis-progression/train.csv\")\npatient_dcm_dict = {}\nroot_dir = \"/kaggle/input/osic-pulmonary-fibrosis-progression/train/\"\nremove_ids = [\"ID00078637202199415319443\", \"ID00105637202208831864134\"]\nfor dirname, _, filenames in os.walk(root_dir):\n    if 'ID' in dirname:\n        dirname_ = dirname.replace(root_dir, \"\")\n        # check if the id is among the ids to be removed\n        s = [i for i in remove_ids if dirname_ == i]\n        if len(s) == 0:            \n            patient_dcm_dict[dirname_] = filenames\n        \npatient_dcm_dict = {k: i for i, k in enumerate(sorted(patient_dcm_dict.keys()))}\n\n","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"afe1d81e-0cdd-4b3e-a436-d23143b144fb","_uuid":"5d1cae34-a9f9-43f7-96c8-d843f70cd282","execution":{"iopub.execute_input":"2020-09-12T09:05:17.049177Z","iopub.status.busy":"2020-09-12T09:05:17.046124Z","iopub.status.idle":"2020-09-12T09:05:24.205944Z","shell.execute_reply":"2020-09-12T09:05:24.204909Z"},"papermill":{"duration":7.316529,"end_time":"2020-09-12T09:05:24.206087","exception":false,"start_time":"2020-09-12T09:05:16.889558","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"dataset = osic_utils.DatasetGen(\n    patient_dcm_dict, \n    split = None, seed = 300, \n    batch_size = 1, \n    root_dir = '/kaggle/input/osic-pulmonary-fibrosis-progression/train/',\n    shuffle = False)\n\ndef add_img_and_stats(path):\n    imgs = [ct.image_preprocess(i) for i in path]\n    #return imgs\n\n    # calculating lung volume\n    lung_vol_out = []\n    for masked_img, st, ps, lps in imgs:\n        lung_vol_out.append(\n        ct.get_volume(masked_img, tf.constant(st), tf.constant(ps), lps)\n        )\n\n    # calculating lung image statistics\n    stats = []\n    for img, _ in lung_vol_out:\n        stats.append(\n            ct.calculate_statistics(tf.constant(img))[None, ...])\n    stats = tf.concat(stats, axis = 0)\n\n    ### return lung volume, mean, variance, skew and kurtosis in that order\n    # divide by the magnitude to decrease range\n    lung_volume = tf.reshape(\n        tf.concat([i[1][None, ...] for i in lung_vol_out], axis = 0) /1e5, \n        [-1, 1])\n    all_feats = tf.cast(\n        tf.concat([lung_volume, stats], axis = 1), \n        dtype = tf.float32)\n    #all_feats = tf.gather(all_feats, idx, axis = 0)\n\n    return tf.constant(imgs[0][0], dtype = 'float32'), all_feats\n\ndataset = dataset.train.map(\n    lambda path, _: tf.py_function(\n        add_img_and_stats, \n        [path], \n        (tf.float32, tf.float32))\n    )\n\nfoo = list(dataset.take(1))\n\n\n\n#foo = add_img_and_stats(\n#    tf.reshape(\n#        tf.constant(\"/kaggle/input/osic-pulmonary-fibrosis-progression/train/ID00007637202177411956430/\"), [-1,]))","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-09-12T09:05:24.400669Z","iopub.status.busy":"2020-09-12T09:05:24.399924Z","iopub.status.idle":"2020-09-12T12:31:45.327932Z","shell.execute_reply":"2020-09-12T12:31:45.328663Z"},"papermill":{"duration":12381.032401,"end_time":"2020-09-12T12:31:45.328875","exception":false,"start_time":"2020-09-12T09:05:24.296474","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"start = time()\nlung_stats = []\n\nfor i, tup in enumerate(dataset):\n    lung_stats.append(tup[1])\n    path = dir_img + '/img_proc_'+ list(patient_dcm_dict.keys())[i] + '.npy.gz'\n    f = gzip.GzipFile(path, \"w\")\n    np.save(file=f, arr=tup[0])\n    f.close()\nend = time()\n\nprint((end - start)/60, 'mins')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-09-12T12:31:45.641090Z","iopub.status.busy":"2020-09-12T12:31:45.640251Z","iopub.status.idle":"2020-09-12T12:31:46.272207Z","shell.execute_reply":"2020-09-12T12:31:46.271272Z"},"papermill":{"duration":0.791044,"end_time":"2020-09-12T12:31:46.272376","exception":false,"start_time":"2020-09-12T12:31:45.481332","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"# prepare data frame for lung image statistics\ndf_stats = pd.DataFrame(tf.concat(lung_stats, axis = 0).numpy(), \n             columns = ['lung_vol', 'mean', 'var', 'skew', 'kurt'])\ndf_stats['PatientId'] = list(patient_dcm_dict.keys())[0:df_stats.shape[0]]\ndf_stats.head()\ndf_stats.to_csv(\"lung_statistics.csv\", index = False)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-09-12T12:31:46.588600Z","iopub.status.busy":"2020-09-12T12:31:46.587793Z","iopub.status.idle":"2020-09-12T12:31:46.591493Z","shell.execute_reply":"2020-09-12T12:31:46.590886Z"},"papermill":{"duration":0.163932,"end_time":"2020-09-12T12:31:46.591639","exception":false,"start_time":"2020-09-12T12:31:46.427707","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"#os.path.getsize('train-image-processed/')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-09-12T12:31:46.901215Z","iopub.status.busy":"2020-09-12T12:31:46.900050Z","iopub.status.idle":"2020-09-12T12:31:46.904164Z","shell.execute_reply":"2020-09-12T12:31:46.904689Z"},"papermill":{"duration":0.160633,"end_time":"2020-09-12T12:31:46.904842","exception":false,"start_time":"2020-09-12T12:31:46.744209","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"# will revisit this image\n\"/kaggle/input/osic-pulmonary-fibrosis-progression/train/ID00105637202208831864134/\"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}