{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Here's a faster way to download images using the principles of concurrency and parallelism. In other words, asynchronous tasking and multiprocessing. It works best on machines with 8+ cores and when downloading large files like videos.\n\nPlease let me know if you have other good methods for downloading large numbers of files!","metadata":{}},{"cell_type":"code","source":"# Libraries for concurrency and parallelism\n! pip install asks trio","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-10-26T19:30:49.476801Z","iopub.execute_input":"2021-10-26T19:30:49.477786Z","iopub.status.idle":"2021-10-26T19:31:02.592136Z","shell.execute_reply.started":"2021-10-26T19:30:49.477681Z","shell.execute_reply":"2021-10-26T19:31:02.591122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pathlib import Path\nimport requests\nfrom os import cpu_count\n\nimport datatable as dt\nimport asks\nimport trio","metadata":{"execution":{"iopub.status.busy":"2021-10-26T19:31:02.594214Z","iopub.execute_input":"2021-10-26T19:31:02.594598Z","iopub.status.idle":"2021-10-26T19:31:02.763612Z","shell.execute_reply.started":"2021-10-26T19:31:02.594549Z","shell.execute_reply":"2021-10-26T19:31:02.762644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cpu_count()","metadata":{"execution":{"iopub.status.busy":"2021-10-26T19:44:46.403247Z","iopub.execute_input":"2021-10-26T19:44:46.403646Z","iopub.status.idle":"2021-10-26T19:44:46.410721Z","shell.execute_reply.started":"2021-10-26T19:44:46.403611Z","shell.execute_reply":"2021-10-26T19:44:46.409776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Path('./pics').mkdir(exist_ok=True)\n\nimg_urls = dt.fread(\"../input/wikipedia-image-caption/test.tsv\", sep='\\t', columns={'image_url'})\nlinks = img_urls.to_list()[0][:1000]","metadata":{"execution":{"iopub.status.busy":"2021-10-26T19:31:02.764884Z","iopub.execute_input":"2021-10-26T19:31:02.765122Z","iopub.status.idle":"2021-10-26T19:31:03.144389Z","shell.execute_reply.started":"2021-10-26T19:31:02.765095Z","shell.execute_reply":"2021-10-26T19:31:03.143492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# fast way\n\nasync def fetch_pic(s, url):\n    r = await s.get(url)\n    return r.content\n\n\nasync def save_pic(s, url):\n    content = await fetch_pic(s, url)\n    filename = f\"pics/{url.split('/')[-1][:100]}\"\n    with open(filename,'wb') as f:\n        f.write(content)\n\n        \nasync def main(links):\n    dname = 'https://upload.wikimedia.org'\n    s = asks.sessions.Session(dname, connections=cpu_count()*2)\n    async with trio.open_nursery() as n:\n        for url in links:\n            n.start_soon(save_pic, s, url)\n\n            \ntrio.run(main, links)","metadata":{"execution":{"iopub.status.busy":"2021-10-26T19:31:03.149097Z","iopub.execute_input":"2021-10-26T19:31:03.149596Z","iopub.status.idle":"2021-10-26T19:31:52.543723Z","shell.execute_reply.started":"2021-10-26T19:31:03.149556Z","shell.execute_reply":"2021-10-26T19:31:52.542751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%bash\nls pics | wc -l\nls pics -U | head -6","metadata":{"execution":{"iopub.status.busy":"2021-10-26T19:31:52.544938Z","iopub.execute_input":"2021-10-26T19:31:52.545188Z","iopub.status.idle":"2021-10-26T19:31:52.581544Z","shell.execute_reply.started":"2021-10-26T19:31:52.545159Z","shell.execute_reply":"2021-10-26T19:31:52.580517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# regular way\n\nfor url in links:\n    r = requests.get(url, stream=True)\n    content=r.content\n    filename = f\"pics/{url.split('/')[-1][:100]}\"\n    with open(filename,'wb') as f:\n        f.write(content)\n    ","metadata":{"execution":{"iopub.status.busy":"2021-10-26T19:31:52.583556Z","iopub.execute_input":"2021-10-26T19:31:52.584482Z","iopub.status.idle":"2021-10-26T19:33:06.001255Z","shell.execute_reply.started":"2021-10-26T19:31:52.584423Z","shell.execute_reply":"2021-10-26T19:33:06.000179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}