{"cells":[{"metadata":{},"cell_type":"markdown","source":"# This code is useful if you download the metadata in your PC\n### **If you use this script, you have to remove some parts. (marked \"### !!!Remove!!! ###\")**\n\n##### This code based on web crawling. So this code is also good to study web crawling or data engineering :)\n---\n"},{"metadata":{"_uuid":"71ef8c0e-135c-4bd2-b36e-d42d1f2ed3b4","_cell_guid":"b86d9ed2-cc33-4439-904a-9e9169ad0c3f","trusted":true},"cell_type":"markdown","source":"# Install and import package"},{"metadata":{"_uuid":"ce17128f-f7b6-4de8-b81e-86c6c423f5a4","_cell_guid":"a2ddc8a4-1e9d-439d-bb74-78a4e2cbf443","trusted":true},"cell_type":"code","source":"!pip install requests bs4 lxml wget","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"46da8b19-b513-405c-91a7-994c14a7431c","_cell_guid":"0bc124e4-fc00-4cc5-9ce5-3415d7a6dac4","trusted":true},"cell_type":"code","source":"import requests\nfrom bs4 import BeautifulSoup\nfrom multiprocessing import Pool\nfrom functools import partial\nimport os\nimport os.path as pth\nimport wget\nfrom zipfile import ZipFile","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c235848c-abe5-4887-8d00-0d0f282d023b","_cell_guid":"f801c243-7dc7-41bc-b2a4-9d44a43f0a9c","trusted":true},"cell_type":"markdown","source":"# Download metadata"},{"metadata":{"_uuid":"49890438-1abe-4a8c-9427-bd07d39133a0","_cell_guid":"b605abed-9b97-485b-bcb7-20ba41c44fb8","trusted":true},"cell_type":"code","source":"def get_metadata(url, base_path='./'):\n    filename = url.split('/')[-1]\n    full_filename = pth.join(base_path, filename)\n    if pth.exists(full_filename):\n        return full_filename, 1\n    wget.download(url, out=base_path)\n    ### If you can't use wget, you can use below blocked code\n    ### But It's much slower than wget...\n    # data = requests.get(url).data\n    # with open(full_filename, 'wb') as f:\n    #     f.write(data)\n    return full_filename, 0","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"709c28c4-c00d-485b-b58c-9be71bc0b0db","_cell_guid":"83cfcdf5-7d30-4ab5-83a2-564ee201e332","trusted":true},"cell_type":"code","source":"target_url = 'https://storage.googleapis.com/openimages/web/download.html'","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bc2b70e5-2293-4611-86b7-957b4e2b5a15","_cell_guid":"fa7974fd-cc7d-4f3d-9990-b26ba6b6ec00","trusted":true},"cell_type":"code","source":"page = requests.get(target_url).text\nsoup = BeautifulSoup(page, 'lxml')\nmain = soup.select_one('div.main')\nrows = main.select('div.row')\nrows = [row for row in rows \n            if row.select_one('div.col-10') and row.select_one('div.col-2.titlecol')]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f997d223-4f3a-4617-b3fb-c04f28a43f47","_cell_guid":"ed72904f-b433-4c5a-9bf5-990c500d6f35","trusted":true},"cell_type":"markdown","source":"### Download data"},{"metadata":{"_uuid":"e02c9cd7-9b7b-4f12-b035-6119c29dcd7a","_cell_guid":"67855444-0b2a-4628-9eb6-f77c72e91150","trusted":true},"cell_type":"code","source":"base_path = 'metadata/'\nos.makedirs(base_path, exist_ok=True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"---\n# !!!Remove!!! ↓"},{"metadata":{"trusted":true},"cell_type":"code","source":"### !!!Remove!!! ###\n### Because of kaggle kernel disk space limit. \n### If you use this script, you must remove below line.\nrows = rows[:10] ### this line or cell must be removed!","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"---"},{"metadata":{"_uuid":"6d27d373-9b38-4934-be15-5f3a93f5a0e5","_cell_guid":"3e5ebacc-781a-4751-a8b3-f40c3f99f4dd","trusted":true},"cell_type":"code","source":"for row in rows:\n    sub_name = row.select_one('div.col-2.titlecol').get_text().strip()\n    if sub_name:\n        sub_path = pth.join(base_path, sub_name)\n        os.makedirs(sub_path, exist_ok=True)\n        hrefs = [a.get('href') for a in row.select('a') if a.get('href')]\n        hrefs = [href for href in hrefs \n                    if href.endswith('.csv') or href.endswith('.txt')]      \n        if hrefs:\n            download_func = partial(get_metadata, base_path=sub_path)\n            pool = Pool(8)\n            for filename, status in pool.imap_unordered(download_func, hrefs):\n                if status == 0:\n                    print(filename + ' is saved.')\n                elif status == 1:\n                    print(filename + ' is already exist.')\n                else:\n                    print('???')\n                \n                ### !!!Remove!!! ###\n                ### Because of kaggle kernel disk space limit. \n                ### If you use this script, you must remove below line.\n                os.remove(filename) ### this line must be removed!\n            \n            pool.close()\n            pool.join()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e9a6416f-836c-4527-85d2-4feba75e9cc3","_cell_guid":"b00dcc85-4934-41d4-b5f5-391530926efe","trusted":true},"cell_type":"markdown","source":"# Download and Extract segmentation image file"},{"metadata":{"_uuid":"20912329-83a8-4df3-acba-47d42fecfbf2","_cell_guid":"45446e80-4c73-4f73-b412-a6f5047fce14","trusted":true},"cell_type":"code","source":"base_path = pth.join('metadata', 'Segmentations')\nos.makedirs(base_path, exist_ok=True)\nbase_url = 'https://storage.googleapis.com/openimages/v5/{}-masks/{}-masks-{}.zip'","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0a845f6c-8a2c-4bf6-a994-1d0e8ea18e8b","_cell_guid":"9c548ba3-6ddb-43e8-ad61-eaf9180242a5","trusted":true},"cell_type":"markdown","source":"### Download zip"},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_list = ['train', 'validation', 'test']","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"---\n# !!!Remove!!! ↓"},{"metadata":{"trusted":true},"cell_type":"code","source":"### !!!Remove!!! ###\n### Because of kaggle kernel disk space limit. \n### If you use this script, you must remove below line.\nsub_list = ['validation', 'test'] ### this line or cell must be removed!","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"---"},{"metadata":{"_uuid":"497f68a7-2706-4d73-ac98-db60c89b6036","_cell_guid":"e8c4ca7e-cd49-4ff8-860a-e5a9561dc4ce","trusted":true},"cell_type":"code","source":"for sub_name in sub_list:\n    sub_path = pth.join(base_path, sub_name)\n    os.makedirs(sub_path, exist_ok=True)\n    urls = [base_url.format(sub_name, sub_name, offset) \n            for offset in list(range(10))+['a','b','c','d','e','f']]\n    download_func = partial(get_metadata, base_path=sub_path)\n    pool = Pool(8)\n    for filename, status in pool.imap_unordered(download_func, urls):\n        if status == 0:\n            print(filename + ' is saved.')\n        elif status == 1:\n            print(filename + ' is already exist.')\n        else:\n            print('???')\n    pool.close()\n    pool.join()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f48301b4-c34b-4a6b-8eb2-83046fd73f5d","_cell_guid":"898f017c-318c-4419-86d8-08a723b16502","trusted":true},"cell_type":"markdown","source":"### Extract image"},{"metadata":{"_uuid":"d6397865-3c1b-41fc-b049-4979776c9fdc","_cell_guid":"4376c4bd-f914-4dd2-9187-19c84b1179d8","trusted":true},"cell_type":"code","source":"for sub_name in sub_list:\n    sub_path = pth.join(base_path, sub_name)\n    zip_filename_list = [filename for filename in os.listdir(sub_path) if filename.endswith('.zip')]\n    for filename in zip_filename_list:\n        segment_zip_filename = pth.join(sub_path, filename)\n        with ZipFile(segment_zip_filename, 'r') as zip_ref:\n            zip_ref.extractall(sub_path)\n        os.remove(segment_zip_filename)\n        print(segment_zip_filename+ ' was extracted')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!ls metadata","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!ls metadata/Segmentations/test | head","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"---\n# !!!Remove!!! ↓"},{"metadata":{"trusted":true},"cell_type":"code","source":"### !!!Remove!!! ###\n### Because of kaggle kernel disk space limit. \n### If you use this script, you must remove below line.\n!rm -rf metadata/ ### this line or cell must be removed!","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"---"}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}