{"cells":[
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output.\n\n"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "\ndef load(base_directory=None):\n    import os\n    basedir = base_directory or os.path.dirname(os.path.abspath(__file__))\n    train_d = pd.read_csv(os.path.join(basedir, 'train.csv'))\n    train_to_biz_id_data = pd.read_csv(os.path.join(basedir, 'train_photo_to_biz_ids.csv'))\n    X_TRAIN = pd.merge(train_d, train_to_biz_id_data, on='business_id')\n    Y_TRAIN = X_TRAIN['labels'].str.get_dummies(sep=' ')#this is much faster than apply\n    del(X_TRAIN['labels'])\n\n    X_TEST = pd.read_csv(os.path.join(basedir, 'test_photo_to_biz.csv'))\n\n    data = {\n        'X_TRAIN' : X_TRAIN,\n        'Y_TRAIN' : Y_TRAIN,\n        'X_TEST' : X_TEST\n    }\n    return data\n\ndata = load('../input')\n"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "data['X_TRAIN'].head()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "data['Y_TRAIN'].head()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "data['X_TEST'].head()"
 }
],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}}, "nbformat": 4, "nbformat_minor": 0}