{"nbformat": 4, "cells": [{"cell_type": "code", "execution_count": null, "outputs": [], "source": ["# This Python 3 environment comes with many helpful analytics libraries installed\n", "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n", "# For example, here's several helpful packages to load in \n", "\n", "import numpy as np # linear algebra\n", "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n", "\n", "# Input data files are available in the \"../input/\" directory.\n", "# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n", "\n", "from subprocess import check_output\n", "print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n", "\n", "# Any results you write to the current directory are saved as output.\n", "import bson\n", "import os\n", "import collections\n", "from tqdm import tqdm_notebook"], "metadata": {"_uuid": "36952d0d346fd8358d0668fa62364c6f436950e2", "collapsed": true, "_cell_guid": "1a7bf7d2-9b0c-45db-a448-08d1cb8b9919"}}, {"cell_type": "markdown", "source": ["This kernel is the work of Bruno do Amaral: https://www.kaggle.com/bguberfain/not-so-naive-way-to-convert-bson-to-files\n", "\n", "I only added the validation folders and a sequential split of the data.\n", "\n", "The resulting train and validation folders take some 85 GiB. "], "metadata": {"_uuid": "02ff69a3ee3b3306823f3f2cf0bbfa8056a94eca", "_cell_guid": "ff6cfb97-395a-447c-ba85-90b948f74c5c"}}, {"cell_type": "code", "execution_count": null, "outputs": [], "source": ["train_folder      = '../output/train'\n", "validation_folder = '../output/validation'\n", "\n", "# Create train folder\n", "if not os.path.exists(train_folder):\n", "    os.makedirs(train_folder)\n", "    \n", "# Create validation folder\n", "if not os.path.exists(validation_folder):\n", "    os.makedirs(validation_folder)\n", "    "], "metadata": {"_uuid": "33e85ee5d4364046c3d1cbfbebbbe8fe3aa8a9a1", "collapsed": true, "_kg_hide-output": true, "_cell_guid": "386bb675-bd96-47c1-8460-6419d185ca10"}}, {"cell_type": "code", "execution_count": null, "outputs": [], "source": ["# Create categories folders\n", "categories = pd.read_csv('../input/category_names.csv', index_col='category_id')\n", "\n", "for category in tqdm_notebook(categories.index):\n", "    os.mkdir(os.path.join(train_folder, str(category)))\n", "    os.mkdir(os.path.join(validation_folder, str(category)))\n", "    "], "metadata": {"_uuid": "8334e62241b7ffc833919c1fcb7818e0a433bd32", "collapsed": true, "_kg_hide-output": true, "_cell_guid": "92085db0-baaf-4cf4-8b09-01e1dbba377d"}}, {"cell_type": "code", "execution_count": null, "outputs": [], "source": ["num_products = 7069896  # 7069896 for train and 1768182 for test\n", "num_prod_train = num_products*0.8   #set 80% of the data as the training set. Leave the remainder as validation set\n", "print('training set will have ', num_prod_train, 'items')\n", "\n", "bar = tqdm_notebook(total=num_products)\n", "counter = 0\n", "with open('../input/train.bson', 'rb') as fbson:\n", "\n", "    data = bson.decode_file_iter(fbson)\n", "    \n", "    for c, d in enumerate(data):\n", "        category = d['category_id']\n", "        _id = d['_id']\n", "        counter += 1\n", "\n", "        for e, pic in enumerate(d['imgs']):\n", "            if counter < num_prod_train :\n", "                fname = os.path.join(train_folder, str(category), '{}-{}.jpg'.format(_id, e))                \n", "            else:\n", "                fname = os.path.join(validation_folder, str(category), '{}-{}.jpg'.format(_id, e))\n", "            with open(fname, 'wb') as f:\n", "                f.write(pic['picture'])\n", "\n", "        bar.update()"], "metadata": {"_uuid": "3d03121f45d0dbb0de4e321edc64c5de9e28e3ee", "collapsed": true, "_kg_hide-output": true, "_cell_guid": "ca0447bf-c152-4392-b6c2-d6fb3fbf30d1"}}, {"cell_type": "code", "execution_count": null, "outputs": [], "source": [], "metadata": {"_uuid": "cc5e54b13a68be031b1c6545e652054c1b9968ee", "collapsed": true, "_cell_guid": "5052e1a2-22e4-4195-a110-1085fe258c15"}}], "metadata": {"kernelspec": {"display_name": "Python 3", "name": "python3", "language": "python"}, "language_info": {"mimetype": "text/x-python", "name": "python", "version": "3.6.1", "pygments_lexer": "ipython3", "codemirror_mode": {"version": 3, "name": "ipython"}, "nbconvert_exporter": "python", "file_extension": ".py"}}, "nbformat_minor": 1}