{
  "cells": [
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "5052d505-c523-4bdc-9349-c2a14b721979"
      },
      "outputs": [],
      "source": [
        "# This Python 3 environment comes with many helpful analytics libraries installed\n",
        "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n",
        "# For example, here's several helpful packages to load in \n",
        "\n",
        "import numpy as np # linear algebra\n",
        "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n",
        "\n",
        "# Input data files are available in the \"../input/\" directory.\n",
        "# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n",
        "\n",
        "from subprocess import check_output\n",
        "print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n",
        "\n",
        "# Any results you write to the current directory are saved as output."
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "6265d833-83fc-3c13-0fd7-5d98e7411212"
      },
      "outputs": [],
      "source": [
        "clicked = pd.read_csv(\"../input/clicks_train.csv\")\n",
        "print(clicked.head())\n",
        "print(\"clicks_train Dimension:\",clicked.shape)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "51ab74c9-92aa-ebf3-6bd7-0e925f16b4be"
      },
      "outputs": [],
      "source": [
        "clicked_1 = clicked[[\"ad_id\",\"clicked\"]].groupby(\"ad_id\").sum()\n",
        "clicked_1 = clicked_1.sort(\"clicked\",ascending=False)\n",
        "clicked_1.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "3a4885a2-e779-7a8a-ed0c-d734e628244a"
      },
      "outputs": [],
      "source": [
        "clicked_2 = pd.DataFrame(clicked[[\"ad_id\"]].groupby(\"ad_id\").size().rename('counts'))\n",
        "clicked_2 = clicked_2.sort(\"counts\",ascending=False)\n",
        "clicked_2.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "70aa7592-f941-8eba-cd5d-a4cb7ecbf67c"
      },
      "outputs": [],
      "source": [
        "click_ana = pd.concat([clicked_1,clicked_2],axis=1)\n",
        "print(click_ana.shape)\n",
        "print(click_ana.head())"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "41e0aea7-0e77-ae8f-506e-10209c71d396"
      },
      "source": [
        "We have made a table contains important information about the ad_id."
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "6876359d-6bd0-cff6-e246-f1c70523504c"
      },
      "outputs": [],
      "source": [
        "del clicked_1, clicked_2"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "7d2d69ec-12e7-ebc4-5d4d-904e1b199692"
      },
      "outputs": [],
      "source": [
        "click_ana[\"rate\"] = click_ana[\"clicked\"]/click_ana[\"counts\"]"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "663223f1-6d79-4e0d-6497-2963f8c709a3"
      },
      "outputs": [],
      "source": [
        "click_ana.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "e8cc18c8-eda8-b56f-1605-d5bc92dec3c6"
      },
      "outputs": [],
      "source": [
        "click_ana.describe()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f91ed105-8d80-cedf-072e-3312d4b17c17"
      },
      "outputs": [],
      "source": [
        "import matplotlib.pyplot as plt\n",
        "from matplotlib.ticker import NullFormatter\n",
        "%matplotlib inline\n",
        "x = list(click_ana[\"counts\"])\n",
        "y = list(click_ana[\"clicked\"])\n",
        "MX = max(x)\n",
        "MY = max(y)\n",
        "plt.figure(figsize=(12, 12))\n",
        "plt.scatter(x,y,color=\"blue\")\n",
        "plt.xlabel(\"# of counts\")\n",
        "plt.ylabel(\"# of cliced\")\n",
        "plt.xlim(0,MX+2)\n",
        "plt.ylim(0,MY+2)\n",
        "plt.show()"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "5b4d9e11-273d-7106-7d60-17df3838c2b9"
      },
      "source": [
        "It may be useful for designing basic rules."
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.5.2"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}