{"cells":[
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output."
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "import numpy as np\nimport os\nimport sys\nimport tensorflow as tf\nfrom PIL import Image\nimport csv"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "def id_loader(file_link):\n    table=dict()\n    table['photo_id']=list()\n    table['business_id']=list()\n    with open(file_link) as csvfile:\n        reader = csv.DictReader(csvfile)\n        for row in reader:\n            table['photo_id'].append(row[\"photo_id\"])\n            table['business_id'].append(row[\"business_id\"])\n    return table"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "def label_loader(file_link):\n    table=dict()\n    table['business_id']=list()\n    table['labels']=list()\n    with open(file_link) as csvfile:\n        reader = csv.DictReader(csvfile)\n        for row in reader:\n            table['business_id'].append(row[\"business_id\"])\n            table['labels'].append(row[\"labels\"])\n    return table"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "id_label=label_loader('../input/train.csv')"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "id_id=id_loader('../input/train_photo_to_biz_ids.csv')"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "def load_jpg(folder, id_label, id_id):\n    tmplist=list()\n    x_size = 500\n    y_size = 375\n    my_size = x_size, y_size\n    pixel=3\n    image_files = os.listdir(folder)\n    dataset = np.ndarray(shape=(len(image_files), y_size, x_size, pixel), dtype=np.float32)\n    label = np.ndarray(shape=(len(image_files), 9), dtype=np.int32)\n    image_index = 0\n    for image in os.listdir(folder):\n        if image.startswith(\".\"):\n            continue\n        templst=image.split(\".\")\n        bzid=id_id[\"business_id\"][id_id[\"photo_id\"].index(templst[0])]\n        label_str=id_label[\"labels\"][id_label[\"business_id\"].index(bzid)]\n        label_list=label_str.split(\" \")\n        for i in range(len(label_list)):\n            try:\n                index=int(label_list[i])\n                label[image_index][index]=1\n            except:\n                label[image_index][0]=0\n        #print(image_index)\n        image_file = os.path.join(folder, image)\n        #print(image_file)\n        im = Image.open(image_file)\n        if im.width < im.height:\n            size=im.height,im.width\n            im=im.rotate(90,expand=1).resize(size)\n        if im.width is not 500 or im.height is not 375:\n            im=im.resize(my_size, Image.ANTIALIAS)\n        dataset[image_index, :, :,:] = im\n        image_index += 1\n    num_images = image_index\n    dataset = dataset[0:num_images, :, :,:]\n    return dataset, label"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "folder = '../input/train_photos/'\ntrain_data, label=load_jpg(folder, id_label, id_id)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": ""
 }
],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}}, "nbformat": 4, "nbformat_minor": 0}