{"metadata": {"kernelspec": {"language": "python", "name": "python3", "display_name": "Python 3"}, "language_info": {"codemirror_mode": {"version": 3, "name": "ipython"}, "nbconvert_exporter": "python", "pygments_lexer": "ipython3", "name": "python", "mimetype": "text/x-python", "file_extension": ".py", "version": "3.6.3"}}, "nbformat_minor": 1, "cells": [{"metadata": {}, "cell_type": "markdown", "source": ["This notebook works towards converting the weird sorted file of tarin_example.bson into diffulet csv dataset where the first column is the category id and the rest  column from 1 till column (180x180x3) \"the number of pixel in each image\" is setted as the pixel value of the affilated image. "]}, {"source": ["# This Python 3 environment comes with many helpful analytics libraries installed\n", "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n", "# For example, here's several helpful packages to load in \n", "\n", "import numpy as np # linear algebra\n", "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n", "import bson\n", "import io\n", "from skimage.data import imread\n", "\n", "# Input data files are available in the \"../input/\" directory.\n", "# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n", "\n", "from subprocess import check_output\n", "data = bson.decode_file_iter(open('../input/train_example.bson', 'rb'))\n", "\n", "df=pd.DataFrame({\"category\":[]})\n", "df=pd.concat([df,pd.DataFrame(columns=[i for i in range(180*180*3)])])\n", "df['category'].astype('object')\n", "index=0\n", "pd.options.display.precision=11\n", "\n", "for i,j in enumerate(data):\n", "    for k in range(len(j['imgs'])):\n", "        image=np.reshape((imread(io.BytesIO(j['imgs'][k]['picture']))),-1)\n", "        image=image.tolist()\n", "        image.insert(0,j[\"category_id\"])\n", "        df.loc[index]=image\n", "        index=index+1    \n", "            \n", "        \n", "df.to_csv(\"train_example.csv\",index=False)\n", "# Any results you write to the current directory are saved as output."], "metadata": {"_uuid": "756f2c6a166a8a66a5ac28cf18620a614756aad7", "_cell_guid": "567f5ef8-a3a6-412c-9100-b729d3860d33"}, "cell_type": "code", "execution_count": null, "outputs": []}, {"metadata": {}, "cell_type": "markdown", "source": ["Now you have a typical dataset where the first columns is the category id and the reset columns are the affilated featuers as shown in the following output."]}, {"source": ["print(df.head(10))"], "metadata": {}, "cell_type": "code", "execution_count": null, "outputs": []}], "nbformat": 4}