{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom multiprocessing import Pool\nfrom tqdm import *\nimport zipfile\n\n\nPATH = '../input/landmark-recognition-2020/train/'\nIMG_SIZE = 512\n\n\ndef zip_and_remove(path):\n    ziph = zipfile.ZipFile(f'{path}.zip','w',zipfile.ZIP_DEFLATED)\n    \n    for root,dirs,files in os.walk(path):\n        print(\"root:\" +root+ \"dirs:\" +dirs+ \"files:\" +files)\n        for file in tqdm(files):\n            file_path = os.path.join(root,file)\n            ziph.write(file_path)\n            os.remove(file_path)\n            \n    ziph.close()\n    \ndef img_proc(ids):\n    path = os.path.join(PATH,ids[0],ids[1],ids[2],ids + '.jpg')\n    img = cv2.resize(cv2.imread(path),(IMG_SIZE,IMG_SIZE))\n    cv2.imwrite('train_img/' + ids + '.jpg',img)\n    \ndef imap_unordered_bar(func,args,n_processes: int=64):\n    p = Pool(n_processes,maxtasksperchild=100)\n    res_list = []\n    with tqdm(total=len(args)) as pbar:\n        for i, res in tqdm(enumerate(p.imap_unordered(func,args))):\n            pbar.update()\n            res_list.append(res)\n            \n    pbar.close()\n    p.close()\n    p.join()\n    return None\n\ndef main():\n    train_df = pd.read_csv(\"../input/landmark-recognition-2020/train.csv\")\n    os.makedirs('train_img_1')\n    tqdm.pandas('Image processing progress')\n    _ = imap_unordered_bar(img_proc,train_df.id.values[10*150_000:])\n    zip_and_remove('train_img')\n    \nif __name__ == '__main__':\n    main()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}