{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"print(os.walk('/kaggle/working'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import io\nimport os\nimport bson\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nfrom imageio import imread\nimport multiprocessing as mp\nfrom glob import iglob","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bson_file = '/kaggle/input/cdiscount-image-classification-challenge/train.bson'\ninput_dir = os.path.abspath(os.path.join(os.getcwd(), ''))\nbase_dir = os.path.join(os.getcwd())\nprint(base_dir)\nimages_dir = os.path.join(base_dir, 'images')\nbson_file = os.path.join(input_dir, bson_file)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"NCORE =  16\nmax_images = 2000\n\nprod_to_category = mp.Manager().dict() # note the difference\n\ndef process(q, iolock):\n    while True:\n        d = q.get()\n        if d is None:\n            break\n        product_id = str(d['_id'])\n        category_id = str(d['category_id'])\n        category_dir = os.path.join(images_dir, category_id)\n        if not os.path.exists(category_dir):\n            #category_count += 1\n            try:\n                os.makedirs(category_dir)\n            except:\n                pass\n\n        prod_to_category[product_id] = category_id\n        for e, pic in enumerate(d['imgs']):\n            picture = imread(io.BytesIO(pic['picture']))\n            picture_file = os.path.join(category_dir, product_id + '_' + str(e) + '.jpg')\n\n            if not os.path.isfile(picture_file):\n                #cv2.imwrite(picture_file,picture)\n\n                plt.imsave(picture_file, picture)\n            \n    \nq = mp.Queue(maxsize=NCORE)\niolock = mp.Lock()\npool = mp.Pool(NCORE, initializer=process, initargs=(q, iolock))\n\n# process the file\n\ndata = bson.decode_file_iter(open(bson_file, 'rb'))\nfor c, d in enumerate(data):\n    if (c+1) >max_images:\n        break\n    q.put(d)  # blocks until q below its max size\n\n    # tell workers we're done\n\nfor _ in range(NCORE):  \n    q.put(None)\npool.close()\npool.join()\n\nprint('Images saved at %s' % images_dir)\n\n# convert back to normal dictionary\nprod_to_category = dict(prod_to_category)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"prod_to_category = pd.DataFrame.from_dict(prod_to_category, orient='index')\nprod_to_category.index.name = '_id'\nprod_to_category.rename(columns={0: 'category_id'}, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"prod_to_category.head()\n\nprod_to_category.to_csv(\"categories.csv\")","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}