{
  "cells": [
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "b77f0cdf-49d7-4c75-4216-855b50287257"
      },
      "source": [
        "Intro\n",
        "-----\n",
        "\n",
        "Every mystery starts with asking some questions, and then trying to reveal the answers. These answers are the key to achieve better results. So, before getting into the predictions part, we are going to unleash the hidden secrets immersed within the data by analyzing and visualize it.\n",
        "\n",
        "So, let's get started!"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "5c8d4d1c-3ace-a432-5b1a-19806be64698"
      },
      "outputs": [],
      "source": [
        "# Imports\n",
        "\n",
        "# pandas\n",
        "import pandas as pd\n",
        "from pandas import Series,DataFrame\n",
        "\n",
        "# numpy, matplotlib, seaborn\n",
        "import numpy as np\n",
        "import matplotlib.pyplot as plt\n",
        "import seaborn as sns\n",
        "sns.set_style('whitegrid')\n",
        "%matplotlib inline"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "1a5c941e-51c3-a047-c700-d10c855d022e"
      },
      "outputs": [],
      "source": [
        "# get clicks_train, clicks_test, & events csv files as a DataFrame\n",
        "clicks_train = pd.read_csv('../input/clicks_train.csv')\n",
        "clicks_test  = pd.read_csv('../input/clicks_test.csv')\n",
        "events_df    = pd.read_csv('../input/events.csv', usecols=['uuid', 'platform', 'geo_location'])\n",
        "# ads_df          = pd.read_csv('../input/promoted_content.csv')\n",
        "# documents_df    = pd.read_csv('../input/documents_meta.csv')\n",
        "# categories_df   = pd.read_csv('../input/documents_categories.csv')\n",
        "\n",
        "\n",
        "# preview the data\n",
        "clicks_train.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "e4e18324-8563-e1dc-4dfc-2d68de19093f"
      },
      "outputs": [],
      "source": [
        "clicks_train.info()\n",
        "print(\"----------------------------\")\n",
        "clicks_test.info()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "4ca1a6a0-f951-72b5-3e53-67ab3abb4028"
      },
      "outputs": [],
      "source": [
        "# Ads\n",
        "\n",
        "# What's the frequency Vs the mean of each ad\n",
        "\n",
        "# The Frequency(Count of occurrence for each value)\n",
        "ads_freq = clicks_train['ad_id'].value_counts()\n",
        "\n",
        "# The mean(The average of clicks)\n",
        "ads_clicked = clicks_train[clicks_train['clicked'] == 1]['ad_id'].value_counts()\n",
        "ads_average = ads_clicked.values / ads_freq[ads_clicked.index]"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "1bf778cd-3d41-9f32-89eb-296ba69bfcf0"
      },
      "outputs": [],
      "source": [
        "# Given the number of clicks for each ad\n",
        "# We can show the important, the max and the min values\n",
        "# This gives a clue about how values(count of clicks) is distributed\n",
        "# For me, I would guess probably it's normally distributed, but, let's see\n",
        "\n",
        "# Plot max, min values, & 2nd, 3rd quartile\n",
        "fig, (axis1) = plt.subplots(1,1,figsize=(12,5))\n",
        "sns.boxplot([ads_clicked], ax=axis1)\n",
        "axis1.set(xlabel='Frequency for count of clicks')"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "d1b1db8d-3345-2492-784a-0e0f6ab91d64"
      },
      "outputs": [],
      "source": [
        "# Huummm, It seems most of the ad clicks lies between 1 and 10000, \n",
        "# and few of them lies after 10000, Isn't it?\n",
        "# But, this doesn't clearly show the frequency for ad clicks\n",
        "# So, Let's get deeper ...\n",
        "\n",
        "# Plot frequency for clicks on ads\n",
        "\n",
        "# And because there are many values(small) that just appeared a few times, \n",
        "# and few values(large) that appeared so much,\n",
        "# Thus, we had to use Log to show all of them.\n",
        "fig, (axis1, axis2) = plt.subplots(2,1,figsize=(12,8))\n",
        "ads_clicked.plot(kind='hist',bins=50,log=True,ax=axis1)\n",
        "axis1.set(ylabel='Log10(Frequency)', xlabel='Count of Clicks')\n",
        "\n",
        "# Plot the average of clicks and the standard deviation\n",
        "# This is a huge std!. According to the empirical rule (given mean=66 & std=578):\n",
        "# 68% of the clicks where between 66 - 578 and 66 + 578 clicks.\n",
        "# 95% of the clicks where between 66 - 2 X (578) and 66 + 2 X (578) clicks.\n",
        "# 98% of the clicks where between 66 - 3 X (578) and 66 + 3 X (578) clicks.\n",
        "Series(ads_clicked.mean()).plot(yerr=ads_clicked.std(),kind='bar',legend=False, ax=axis2)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "82a3348e-5494-c549-b676-a97949b69ec2"
      },
      "outputs": [],
      "source": [
        "# Now, we can also dive deeper, \n",
        "# and see the the percentage of ads that were clicked less than(or equal) X times?\n",
        "\n",
        "ads_perc = Series()\n",
        "for i in [2, 10, 50, 100, 1000, 5000]:\n",
        "    ads_perc[str(i)] = round((ads_clicked.values <= i).mean() * 100, 2)\n",
        "\n",
        "ax = ads_perc.plot(kind='bar', figsize=(12,3), colormap=\"summer\")\n",
        "ax = ax.set(ylabel='Percentage', xlabel='Count of Clicks')"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "2fd97e83-f567-6ae0-704b-3fe4b975fdd5"
      },
      "outputs": [],
      "source": [
        "# Finally, it's time to show the actual clicks Vs views\n",
        "# The frequency for count of clicks stops at almost 45000 click, \n",
        "# while there is a good presence for the freqeuncy for count of views after.\n",
        "\n",
        "fig, (axis1) = plt.subplots(1,1,figsize=(12,5))\n",
        "\n",
        "ads_clicked.name = 'Frequency for count of clicks'\n",
        "ads_freq.name =  'Frequency for count of views'\n",
        "\n",
        "ads_clicked.plot(kind='hist',bins=50,normed=True,log=True,color='indianred',alpha=0.5,legend=True)\n",
        "ads_freq.plot(kind='hist',bins=50,normed=True,log=True,alpha=0.5,legend=True)\n",
        "axis1.set(ylabel='Log10(Frequency)', xlabel='Count of Clicks')"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "cf497aa0-0154-b513-48a3-f4153a279422"
      },
      "outputs": [],
      "source": [
        "# Users\n",
        "# How about users? What's the frequency for user clicks? \n",
        "# Is it going to be just like clicks on ad?; \n",
        "# where we have many values(small) appeared a few times, and few values(large) that appeared so much\n",
        "users_freq = events_df['uuid'].value_counts()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "21e76f53-a23a-d030-eadd-8b17c2f23f93"
      },
      "outputs": [],
      "source": [
        "# Plot frequency for user clicks\n",
        "\n",
        "fig, (axis1, axis2) = plt.subplots(2,1,figsize=(12,8))\n",
        "\n",
        "# Same thing here; many values(small) appeared a few times, \n",
        "# and few large values(large) that appeared so much\n",
        "users_freq.plot(kind='hist',log=True,colormap=\"Set2\",bins=50,ax=axis1)\n",
        "axis1.set(ylabel='Log10(Frequency)', xlabel='Count of Clicks')\n",
        "\n",
        "# What's the percentage of users who clicked on ads less than(or equal) X times?\n",
        "users_perc = Series()\n",
        "for i in [2, 3, 5, 10, 50]:\n",
        "    users_perc[str(i)] = round((users_freq.values <= i).mean() * 100, 2)\n",
        "\n",
        "users_perc.plot(kind='bar',colormap=\"Set2\",ax=axis2)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "20671810-b489-5b94-efaa-c1bbcdc9bcd0"
      },
      "outputs": [],
      "source": [
        "# Locations\n",
        "# Grap the country from location(country>state>DMA)\n",
        "events_df['geo_location'] = events_df['geo_location'].apply(lambda x: str(x).split(\">\")[0])\n",
        "\n",
        "# How many times each country participated in ad clicks?\n",
        "# Limit the answer to only countries with more than(or equal) 100000 participation\n",
        "location_freq = events_df['geo_location'].value_counts()\n",
        "location_sum  = location_freq.sum()\n",
        "location_freq = location_freq[location_freq >= 100000]\n",
        "\n",
        "fig, (axis1) = plt.subplots(1,figsize=(12,5))\n",
        "location_freq.plot(kind='bar',colormap=\"Set3\",ax=axis1)\n",
        "\n",
        "for p in axis1.patches:\n",
        "        axis1.annotate('%{:.2f}'.format(p.get_height() * 100 / location_sum), (p.get_x()+0.1, p.get_height()+100000))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "64c4b24d-a2a2-9a0c-243a-d7b0ea55294a"
      },
      "outputs": [],
      "source": [
        "# Platform\n",
        "# Just a quick look at different platforms, and see which is more impactful than the other\n",
        "\n",
        "# Make sure all values are consistent; no 1(int) & \"1\"(str) at the same time!\n",
        "events_df[\"platform\"] = events_df[\"platform\"].map({1: \"1\", 2: \"2\", 3: \"3\"})\n",
        "events_df[\"platform\"] = events_df[\"platform\"].astype(str)\n",
        "\n",
        "# Remove all NaN values\n",
        "platform_freq = events_df[\"platform\"].value_counts()\n",
        "del platform_freq['nan']\n",
        "platform_sum = platform_freq.sum()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "9192b5ee-97dc-689d-0f72-94455b9ec3cd"
      },
      "outputs": [],
      "source": [
        "# Plot count(frequency) for every platform\n",
        "fig, (axis1) = plt.subplots(1,figsize=(12,5))\n",
        "platform_freq.plot(kind='bar',colormap=\"Set3\",ax=axis1)\n",
        "\n",
        "for p in axis1.patches:\n",
        "        axis1.annotate('%{:.2f}'.format(p.get_height() * 100 / platform_sum), (p.get_x()+0.1, p.get_height()+100000))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "842fc708-ac97-1948-3f6b-b6ceb56dca5c"
      },
      "outputs": [],
      "source": [
        "# Predictions\n",
        "# The requirements: For every set of recommendation, sort the ads according to their likelyhood of being clicked\n",
        "# The solution: For every set of recommendation, we are going to sort the ads based on one of the following:\n",
        "    # 1. Count of ad Views(clicked or not clicked)\n",
        "    # 2. Count of ad Clicks\n",
        "    # 3. Average of ad Clicks = Count of ad Clicks / Count of ad Views\n",
        "    # 4. Adjusted Average(with constant) = Count of ad Clicks / (Count of ad Views + constant)\n",
        "    # 5. Adjusted Average(using power) = Count of ad Clicks^2 / Count of ad Views\n",
        "    # 6. Log10(Count of ad Clicks)\n",
        "    # 7. Probability Density Function F(ad) = (1/ Max - Min) X (Clicks - Min) \u2014 Max & Min for number of ad Clicks \n",
        "    # 8. Probability Density Function F(ad) = (1/ Max - Min) X (Clicks - Min) \u2014 Max & Min for number of ad Clicks in the current set of recommendation\n",
        "    # 9. Calculate the zscore Count of ad Clicks\n",
        "    # .... \n",
        "    \n",
        "# The first solution can be misleading, as an ad can be have huge number of views but few clicks.\n",
        "# Solutions 2, 6, 7, 8, 9, are almost the same. They depdend on the Count of Clicks for each ad.\n",
        "# The 3rd Solution is reasonable, but, it can be tricky when you have an ad with views=2 and clicks=2,\n",
        "    # and, another ad with views=1000 and clicks=800, \n",
        "    # so, the probability for the first add to be clicked is 100%, while the second is 80%, \n",
        "    # although the second ad has much higher number of clicks.\n",
        "# The 4th and 5th Solutions are almost the same, and they solve the problem of the 3rd solution.\n",
        "    # The 4th solution penalizes ads with small number of clicks by adding a fixed constant(usually the average of number of views)\n",
        "    # The 5th solution powers the count of clicks, which in turn rewards the ads with large number of clicks.\n",
        "    # NOTE: The constant can be tuned to improve the score.\n",
        "    \n",
        "# We will go with the 4th Solution, and see what we will get.\n",
        "\n",
        "# First, clear up memory!\n",
        "import gc\n",
        "try: del clicks_train,clicks_test,events_df\n",
        "except: pass;\n",
        "gc.collect()\n",
        "\n",
        "# Submission\n",
        "constant = int(ads_freq.mean())\n",
        "ads_adj_average = ads_clicked.values / ( ads_freq[ads_clicked.index] + constant ) \n",
        "# ads_adj_average = (ads_clicked.values**2) / ads_freq[ads_clicked.index] \n",
        "\n",
        "def get_score(ad):\n",
        "    if ad not in ads_adj_average:\n",
        "        return 0\n",
        "    return ads_adj_average[ad] \n",
        "\n",
        "def solve(ads):\n",
        "    # convert to int so we can sort\n",
        "    ads = map(int, ads.split())\n",
        "    # sort according to get_score function\n",
        "    ads = sorted(ads, key=get_score, reverse=True) \n",
        "    # convert back to string so we can join by \" \"\n",
        "    return \" \".join(map(str, ads)) \n",
        "   \n",
        "# Q: Why we are using sample_submission.csv file instead of the clicks_test.csv file?\n",
        "# A: The sample_submission.csv files contains the same data as in clicks_test.csv \n",
        "# but grouped by display_id, where each display_id has the ad ids separated by space.\n",
        "\n",
        "submission = pd.read_csv(\"../input/sample_submission.csv\") \n",
        "submission['ad_id'] = submission['ad_id'].apply(lambda ads: solve(ads))\n",
        "\n",
        "submission.to_csv(\"outbrain.csv\", index=False)"
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.5.2"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}