{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Just a really quick demonstration of how to use the glob library\n\ncredit: [\"Indian Pythonista on YT\"](https://www.youtube.com/watch?v=Vc5kGYty18k&ab_channel=IndianPythonista)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import os\nimport glob","metadata":{"execution":{"iopub.status.busy":"2021-08-14T14:05:09.465578Z","iopub.execute_input":"2021-08-14T14:05:09.465983Z","iopub.status.idle":"2021-08-14T14:05:09.470066Z","shell.execute_reply.started":"2021-08-14T14:05:09.465945Z","shell.execute_reply":"2021-08-14T14:05:09.469051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths = os.listdir('../input/rsna-miccai-brain-tumor-radiogenomic-classification')\npaths","metadata":{"execution":{"iopub.status.busy":"2021-08-14T14:05:54.414739Z","iopub.execute_input":"2021-08-14T14:05:54.415356Z","iopub.status.idle":"2021-08-14T14:05:54.428307Z","shell.execute_reply.started":"2021-08-14T14:05:54.415320Z","shell.execute_reply":"2021-08-14T14:05:54.427387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Ordinarily UNIX can be used in the command line in order to find specific files based on tags such as the following case;","metadata":{}},{"cell_type":"code","source":"#create a space to save the files with .csv tag\ncsv_files = [] \n\n#iterate through the paths variable set in the initial step and append any path with the csv tag to the list we just made\nfor path in paths:\n    if path.endswith(\".csv\"):\n        csv_files.append(path)\n        \n#prints out the array\ncsv_files","metadata":{"execution":{"iopub.status.busy":"2021-08-14T14:09:08.499209Z","iopub.execute_input":"2021-08-14T14:09:08.499575Z","iopub.status.idle":"2021-08-14T14:09:08.507912Z","shell.execute_reply.started":"2021-08-14T14:09:08.499545Z","shell.execute_reply":"2021-08-14T14:09:08.506706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> now with glob we can do the same thing, except markedly faster and in a less verbose fashion","metadata":{}},{"cell_type":"code","source":"#remember to have imported glob at this point\n\n# glob.glob('file-location/*desired-tag')\nglob.glob('../input/rsna-miccai-brain-tumor-radiogenomic-classification/*.csv')","metadata":{"execution":{"iopub.status.busy":"2021-08-14T14:11:59.634651Z","iopub.execute_input":"2021-08-14T14:11:59.635065Z","iopub.status.idle":"2021-08-14T14:11:59.647124Z","shell.execute_reply.started":"2021-08-14T14:11:59.635032Z","shell.execute_reply":"2021-08-14T14:11:59.645797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> stepping up the complexity by just a notch, we can use this functionality recursively to grab any type of file from all child directories of a parent","metadata":{}},{"cell_type":"code","source":"#I am going to switch to grabbing the dcm tag because it is more relevant with this dataset \n\n#glob.glob('parent-dir-location/**/*.target-tag')\nglob.glob(\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/test/00001/**/*.dcm\", recursive=True)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> so here,","metadata":{}},{"cell_type":"code","source":"# **/* ","metadata":{"execution":{"iopub.status.busy":"2021-08-14T14:21:52.666164Z","iopub.execute_input":"2021-08-14T14:21:52.666579Z","iopub.status.idle":"2021-08-14T14:21:52.670766Z","shell.execute_reply.started":"2021-08-14T14:21:52.666537Z","shell.execute_reply":"2021-08-14T14:21:52.669649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> indicates that we are grabbing the contents of any file within the parent-dir and then grabbing the target tag referenced by the single astrik","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}