{"nbformat_minor": 1, "metadata": {"kernelspec": {"language": "python", "display_name": "Python 3", "name": "python3"}, "language_info": {"name": "python", "file_extension": ".py", "version": "3.6.3", "nbconvert_exporter": "python", "mimetype": "text/x-python", "codemirror_mode": {"name": "ipython", "version": 3}, "pygments_lexer": "ipython3"}}, "nbformat": 4, "cells": [{"metadata": {"collapsed": true, "_uuid": "89c9df58a57ad15ce0cd4f7ee57bd3891aa0c121", "_cell_guid": "ef99f8bd-98d7-4420-a7ce-0ee480ab30dd"}, "cell_type": "code", "source": ["from __future__ import print_function\n", "\n", "import numpy as np # linear algebra\n", "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n", "\n", "import io\n", "import bson\n", "import matplotlib.pyplot as plt\n", "from skimage.data import imread   # or, whatever image library you prefer\n", "import random"], "execution_count": null, "outputs": []}, {"metadata": {"collapsed": true, "_uuid": "6082495ad40ab4ec4e943371029ac441180c422b", "_cell_guid": "364aa6cb-b336-47fd-8a23-73836070ab1c"}, "cell_type": "code", "source": ["def extract_categories_df(num_images):\n", "    img_category = list()\n", "    item_locs_list = list()\n", "    items_len_list = list()\n", "    pic_ind_list = list()\n", "    prod_id_list = list()\n", "\n", "    with open('../input/train.bson', 'rb') as f:\n", "        data = bson.decode_file_iter(f)\n", "        last_item_loc = 0\n", "        item_len = 0\n", "        for c, d in enumerate(data):\n", "            loc = f.tell()\n", "            item_len = loc - last_item_loc\n", "            category_id = d['category_id']\n", "            prod_id = d[\"_id\"]\n", "\n", "            for e, pic in enumerate(d['imgs']):\n", "                prod_id_list.append(prod_id)\n", "                img_category.append(category_id)\n", "                item_locs_list.append(last_item_loc)\n", "                items_len_list.append(item_len)\n", "                pic_ind_list.append(e)\n", "                \n", "                if num_images is not None:\n", "                    if len(img_category) >= num_images:\n", "                        break\n", "            \n", "            last_item_loc = loc\n", "            \n", "            if num_images is not None:\n", "                if len(img_category) >= num_images:\n", "                    break\n", "    \n", "    f.close()\n", "    df_dict = {\n", "        'category': img_category,\n", "        \"prod_id\": prod_id_list,\n", "        \"img_id\": range(len(img_category)),\n", "        \"item_loc\": item_locs_list,\n", "        \"item_len\": items_len_list,\n", "        \"pic_ind\": pic_ind_list\n", "    }\n", "    df = pd.DataFrame(df_dict)\n", "    df.to_csv(\"all_images_categories.csv\", index=False)\n", "        \n", "    return df\n", "\n", "def get_image(image_id,data_df,fh):\n", "    img_info = data_df[data_df[\"img_id\"] == image_id]\n", "    item_loc = img_info[\"item_loc\"].values[0]\n", "    item_len = img_info[\"item_len\"].values[0]\n", "    pic_ind = img_info[\"pic_ind\"].values[0]\n", "    fh.seek(item_loc)\n", "    item_data = fh.read(item_len)\n", "    d = bson.BSON.decode(item_data)\n", "    \n", "    picture = imread(io.BytesIO(d[\"imgs\"][pic_ind]['picture']))\n", "    return picture"], "execution_count": null, "outputs": []}, {"metadata": {"_uuid": "4db8560f44e6256351881454b612c30b655c85cb", "_cell_guid": "211a0a9f-9796-42ce-b7d1-4261efc741f2"}, "cell_type": "code", "source": ["cat_df = extract_categories_df(None)"], "execution_count": null, "outputs": []}, {"metadata": {"_uuid": "35433907771bb6d58e8897cbe143b2de34655385", "_cell_guid": "c3499230-021e-4d0e-9894-47e72cf87bbb"}, "cell_type": "code", "source": ["print(cat_df.iloc[0])"], "execution_count": null, "outputs": []}, {"metadata": {"collapsed": true, "_uuid": "b60935e41b3ca2f996d9cd9022eb6eefe4220033", "_cell_guid": "d07af531-77d9-4dd1-94a9-149fcd843e45"}, "cell_type": "code", "source": ["train_fh = open('../input/train.bson', 'rb')"], "execution_count": null, "outputs": []}, {"metadata": {"_uuid": "122b69a6670dcb8b2271503c7d0774c11ca07dbe", "_cell_guid": "24185a6f-e31c-481c-9835-71c68add0a4b"}, "cell_type": "code", "source": ["pic = get_image(0,cat_df,train_fh)\n", "plt.imshow(pic);\n", "plt.show()\n", "pic = np.rot90(pic)\n", "plt.imshow(pic);\n", "plt.show()\n", "pic = np.flip(pic,axis=0)\n", "plt.imshow(pic);\n", "plt.show()"], "execution_count": null, "outputs": []}, {"metadata": {"_uuid": "07a492e57f1aabbfb251d5b946ae26e59491230f", "_cell_guid": "d6388383-3c8d-4deb-95ae-d3122bdd627e"}, "cell_type": "code", "source": ["pic = get_image(500,cat_df,train_fh)\n", "plt.imshow(pic);"], "execution_count": null, "outputs": []}, {"metadata": {"_uuid": "7f6e3ede325610d0bcdf9f901bf7cec37ac57b57", "_cell_guid": "3ddca90d-504e-4d72-a22b-5cf4e679f56f"}, "cell_type": "code", "source": ["pic = get_image(20,cat_df,train_fh)\n", "plt.imshow(pic);"], "execution_count": null, "outputs": []}, {"metadata": {"scrolled": false, "_uuid": "a0217d99eb6ea14cbc757a7355cbbd6b2ac44c16", "_cell_guid": "6253d4a6-ff74-4406-8979-0d2c1d0a1476"}, "cell_type": "code", "source": ["for i in random.sample(range(len(cat_df)),20):\n", "    print(i)\n", "    pic = get_image(i,cat_df,train_fh)\n", "    plt.imshow(pic);\n", "    plt.show();"], "execution_count": null, "outputs": []}, {"metadata": {"collapsed": true, "_uuid": "648bd86620709c0bb71346db1f071074493e9d3e", "_cell_guid": "0bceae1c-4d34-4e05-a178-b255394043aa"}, "cell_type": "code", "source": ["def extract_test_df(num_images):\n", "    prod_id_list = list()\n", "    item_locs_list = list()\n", "    items_len_list = list()\n", "    pic_ind_list = list()\n", "\n", "    with open('../input/test.bson', 'rb') as f:\n", "        data = bson.decode_file_iter(f)\n", "        last_item_loc = 0\n", "        item_len = 0\n", "        for c, d in enumerate(data):\n", "            loc = f.tell()\n", "            item_len = loc - last_item_loc\n", "            prod_id = d[\"_id\"]\n", "\n", "            for e, pic in enumerate(d['imgs']):\n", "                prod_id_list.append(prod_id)\n", "                item_locs_list.append(last_item_loc)\n", "                items_len_list.append(item_len)\n", "                pic_ind_list.append(e)\n", "                \n", "                if num_images is not None:\n", "                    if len(prod_id) >= num_images:\n", "                        break\n", "            \n", "            last_item_loc = loc\n", "            \n", "            if num_images is not None:\n", "                if len(prod_id) >= num_images:\n", "                    break\n", "    \n", "    f.close()\n", "    df_dict = {\n", "        'prod_id': prod_id_list,\n", "        \"img_id\": range(len(prod_id_list)),\n", "        \"item_loc\": item_locs_list,\n", "        \"item_len\": items_len_list,\n", "        \"pic_ind\": pic_ind_list\n", "    }\n", "    df = pd.DataFrame(df_dict)\n", "    df.to_csv(\"all_test_images_categories.csv\", index=False)\n", "        \n", "    return df"], "execution_count": null, "outputs": []}, {"metadata": {"collapsed": true}, "cell_type": "code", "source": ["test_cat_df = extract_test_df(None)"], "execution_count": null, "outputs": []}, {"metadata": {"scrolled": false}, "cell_type": "code", "source": ["test_fh = open('../input/test.bson', 'rb')\n", "for i in random.sample(range(len(test_cat_df)),20):\n", "    print(i)\n", "    pic = get_image(i,test_cat_df,test_fh)\n", "    plt.imshow(pic);\n", "    plt.show();"], "execution_count": null, "outputs": []}, {"metadata": {}, "cell_type": "code", "source": ["img_num_train = cat_df[\"pic_ind\"].value_counts()\n", "img_num_train.plot(kind=\"bar\")\n", "plt.show()\n", "img_num_train = cat_df[\"category\"].value_counts()\n", "img_num_train.plot(kind=\"bar\")\n", "plt.show()"], "execution_count": null, "outputs": []}, {"metadata": {}, "cell_type": "code", "source": ["img_num_test = test_cat_df[\"pic_ind\"].value_counts()\n", "img_num_test.plot(kind=\"bar\")\n", "plt.show()"], "execution_count": null, "outputs": []}, {"metadata": {}, "cell_type": "code", "source": ["## Data Statistics\n", "print(\"## Total number of images in train = {:d}\".format(len(cat_df)))\n", "print(\"## Total number of products in train = {:d}\".format(len(pd.unique(cat_df[\"prod_id\"]))))\n", "print(\"## Total number of categories in train = {:d}\".format(len(pd.unique(cat_df[\"category\"]))))\n", "print(\"## Total number of images in test = {:d}\".format(len(test_cat_df)))\n", "print(\"## Total number of products in test = {:d}\".format(len(pd.unique(test_cat_df[\"prod_id\"]))))"], "execution_count": null, "outputs": []}]}