{
  "cells": [
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "ecc3aa83-a935-28eb-15df-c0b81db51afa"
      },
      "source": [
        "view on data"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "be468095-0ac2-3ce3-ebec-c73dc6d330c7"
      },
      "outputs": [],
      "source": [
        "# This Python 3 environment comes with many helpful analytics libraries installed\n",
        "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n",
        "# For example, here's several helpful packages to load in \n",
        "\n",
        "import numpy as np # linear algebra\n",
        "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n",
        "import os\n",
        "\n",
        "# Input data files are available in the \"../input/\" directory.\n",
        "# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n",
        "\n",
        "from subprocess import check_output\n",
        "print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "087edd23-6df8-b252-e367-af68eccb65bf"
      },
      "outputs": [],
      "source": [
        "base_path = '../input'\n",
        "click_train_path = os.path.join(base_path, 'clicks_train.csv')\n",
        "click_test_path = os.path.join(base_path, 'clicks_test.csv')\n",
        "documents_categories_path = os.path.join(base_path, 'documents_categories.csv')\n",
        "documents_entities_path = os.path.join(base_path, 'documents_entities.csv')\n",
        "documents_meta_path = os.path.join(base_path, 'documents_meta.csv')\n",
        "events_path = os.path.join(base_path, 'events.csv')\n",
        "page_views_sample_path = os.path.join(base_path, 'page_views_sample.csv')\n",
        "promoted_content_path = os.path.join(base_path, 'promoted_content.csv')\n",
        "sample_submission_path = os.path.join(base_path, 'sample_submission.csv')\n",
        "\n",
        "print (click_train_path, os.path.exists(click_train_path))\n",
        "print (click_test_path, os.path.exists(click_test_path))\n",
        "print (documents_categories_path, os.path.exists(documents_categories_path))\n",
        "print (documents_entities_path, os.path.exists(documents_entities_path))\n",
        "print (documents_meta_path, os.path.exists(documents_meta_path))\n",
        "print (events_path, os.path.exists(events_path))\n",
        "print (page_views_sample_path, os.path.exists(page_views_sample_path))\n",
        "print (promoted_content_path, os.path.exists(promoted_content_path))\n",
        "print (sample_submission_path, os.path.exists(sample_submission_path))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "c640224d-cede-d8b0-e552-1bc512bd68a9"
      },
      "outputs": [],
      "source": [
        "click_train_df = pd.read_csv(click_train_path)\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "79bd9d41-13f1-1de3-38d8-bbd2e499fb71"
      },
      "outputs": [],
      "source": [
        "#### clicked per each user.\n",
        "unique_display_id = len(click_train_df.display_id.unique())\n",
        "unique_ad_id = len(click_train_df.ad_id.unique())\n",
        "ctr = click_train_df.clicked.sum()/click_train_df.shape[0]\n",
        "\n",
        "print ('unique display id: ', unique_display_id)\n",
        "print ('unique ad id: ', unique_ad_id)\n",
        "print ('click sum: ', click_train_df.clicked.sum())\n",
        "print('ctr:', ctr)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "de0b14dc-8452-bab6-bc62-b420ece92fd0"
      },
      "outputs": [],
      "source": [
        "ctr = click_train_df.clicked.sum()/click_train_df.shape[0]\n",
        "print('ctr:', ctr)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "b8f3d531-7750-9b0d-1e63-217deb31e543"
      },
      "outputs": [],
      "source": ""
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.5.2"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}