{
  "cells": [
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "e08eabe2-018a-1f8c-87bf-6f6896c5c7c5"
      },
      "outputs": [],
      "source": [
        "# This Python 3 environment comes with many helpful analytics libraries installed\n",
        "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n",
        "# For example, here's several helpful packages to load in \n",
        "\n",
        "import numpy as np # linear algebra\n",
        "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n",
        "\n",
        "# Input data files are available in the \"../input/\" directory.\n",
        "# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n",
        "\n",
        "from subprocess import check_output\n",
        "print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n",
        "\n",
        "# Any results you write to the current directory are saved as output.\n",
        "\n",
        "path = \"../input/\"\n",
        "files = ['documents_categories.csv', 'documents_meta.csv', 'documents_entities.csv', 'promoted_content.csv', 'documents_topics.csv', 'events.csv', 'page_views_sample.csv', 'clicks_train.csv']\n",
        "\n",
        "for fname in files:\n",
        "    print('_'*50)\n",
        "    try:        \n",
        "        print(fname, ' reading...')\n",
        "        dtype = None\n",
        "        if fname == 'promoted_content.csv':\n",
        "            df = pd.read_csv(path + fname, dtype = {'advertiser_id': np.str})\n",
        "        elif fname == 'documents_meta.csv':\n",
        "            df = pd.read_csv(path + fname, parse_dates = ['publish_time'])\n",
        "        elif fname == 'documents_topics.csv':\n",
        "            df = pd.read_csv(path + fname, dtype = {'geo_location': np.str})            \n",
        "        elif fname == 'events.csv':\n",
        "            df = pd.read_csv(path + fname, dtype = {'platform': np.str})\n",
        "        else:\n",
        "            df = pd.read_csv(path + fname)\n",
        "        print(df.columns)\n",
        "        print(df.head())\n",
        "        print(df.describe())\n",
        "    except:\n",
        "        print('error in ', fname)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "7e600c38-67f0-0e56-1f93-a84db6ecf8ad"
      },
      "outputs": [],
      "source": [
        ""
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.5.2"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}