{"cells":[{"metadata":{},"cell_type":"markdown","source":"## Generate DICOM Metadata DataFrame with fast.ai v2","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"With much thanks to fast.ai's medical module we can easily generate metadata from DICOM Files with very few  lines of code ","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"Based on this great kernel https://www.kaggle.com/jhoward/creating-a-metadata-dataframe-fastai/notebookgreat kernel by Sir Jeremy Howard","execution_count":null},{"metadata":{"_kg_hide-output":true,"trusted":true},"cell_type":"code","source":"!pip install fastai2==0.0.17","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from fastai2.basics           import *\nfrom fastai2.medical.imaging  import *","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"main_dir=Path(\"../input/osic-pulmonary-fibrosis-progression\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-output":false},"cell_type":"code","source":"files=[]\nfor dirname, _, filenames in os.walk(main_dir):\n    for filename in (filenames):\n        files.append(os.path.join((dirname), filename))        \nfiles=[x for x in files if '.csv' not in x]\ntrain_images= [Path(x) for x in files if 'train'  in x]\ntest_images = [Path(x) for x in files if 'test'   in x]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"To turn the DICOM file metadata into a DataFrame we can use the from_dicoms function that fastai v2 adds. By passing px_summ=True summary statistics of the image pixels (mean/min/max/std) will be added to the DataFrame as well.\n","execution_count":null},{"metadata":{"_kg_hide-output":true,"trusted":true},"cell_type":"code","source":"dcm_metadata_train=pd.DataFrame.from_dicoms(train_images,px_summ=True)\ndcm_metadata_test=pd.DataFrame.from_dicoms(test_images,px_summ=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dcm_metadata_train=dcm_metadata_train.dropna(axis=1)\ndcm_metadata_train=dcm_metadata_train.drop(\"PatientSex\",axis=1)\ndcm_metadata_test=dcm_metadata_test.dropna(axis=1)\ndcm_metadata_test=dcm_metadata_test.drop(\"PatientSex\",axis=1)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dcm_metadata_train.to_csv(\"dcm_metadata_train.csv\")\ndcm_metadata_test.to_csv(\"dcm_metadata_test.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dcm_metadata_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dcm_metadata_test.head()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}