{
  "cells": [
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "7f661eb9-1580-f39c-79ed-f16d35e16732"
      },
      "source": [
        "# Short testing for the data exploring "
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "492df0bd-542f-ddd2-3ccd-afdaa7a7b10d"
      },
      "outputs": [],
      "source": [
        "# This Python 3 environment comes with many helpful analytics libraries installed\n",
        "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n",
        "# For example, here's several helpful packages to load in \n",
        "\n",
        "import numpy as np # linear algebra\n",
        "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n",
        "\n",
        "# Input data files are available in the \"../input/\" directory.\n",
        "# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n",
        "\n",
        "from subprocess import check_output\n",
        "print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n",
        "\n",
        "# Any results you write to the current directory are saved as output."
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "9c02b9e1-1882-a60b-7f5b-ce44518aecc2"
      },
      "source": [
        "## clicks_test.csv, clicks_train.csv"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "2b52a4ec-17d8-381e-fc0d-d76b1eeea4c6"
      },
      "outputs": [],
      "source": [
        "# only contain the display_id and ad_id. \n",
        "cli_test_demo = pd.read_csv('../input/clicks_test.csv', nrows=10)\n",
        "cli_test_demo"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "d9e0496e-3dc4-a32d-f57a-057188a4c906"
      },
      "outputs": [],
      "source": [
        "# the clicks_train file contain more information like the clicked item\n",
        "cli_train_demo = pd.read_csv('../input/clicks_train.csv', nrows=10)\n",
        "cli_train_demo"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "6dc04e33-4639-9f72-4ffa-086e1ad9ebed"
      },
      "source": [
        "## documents information"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "604f7b41-e78e-b2ea-203d-1b34dd0a4308"
      },
      "outputs": [],
      "source": [
        "# note here is the confidence_level. By this way we can decrease the vector length by projection the document id to category id\n",
        "\n",
        "do_ca = pd.read_csv('../input/documents_categories.csv', nrows=10)\n",
        "do_ca"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "7e862209-7488-6713-090d-2296a154dd20"
      },
      "outputs": [],
      "source": [
        "#Don't quite understand about this file\n",
        "#documents_entities.csv give the confidence that the given entity was referred to in the document.\n",
        "#an entity_id can represent a person, organization, or location\n",
        "#Maybe useless than the categories files\n",
        "do_en = pd.read_csv('../input/documents_entities.csv', nrows=10)\n",
        "do_en\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "4be01e1d-b5b7-1319-535a-43e2ed49f7d9"
      },
      "outputs": [],
      "source": [
        "#Could it decrease the length of the document vector? I don't think so.\n",
        "#which will make more important for a document, its content or who publish it? its content\n",
        "do_me = pd.read_csv('../input/documents_meta.csv', nrows=10)\n",
        "do_me\n",
        "#We may ignore this file\n",
        "#Or we may include the source_id"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "9f8b9726-59d3-672e-d58c-1382c90cf7da"
      },
      "outputs": [],
      "source": [
        "#the same format as the categories and entities file\n",
        "do_to = pd.read_csv('../input/documents_topics.csv', nrows=10)\n",
        "do_to"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "7a198840-7dd9-d12c-f287-974636aff501"
      },
      "source": [
        "Four files above are mentioned the information about the documents. We could represent the document_id as its category, entity, and topic. However, sometimes the confidence_level are very very low which indict the outbrain could not find a most likely class for it.   "
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "7fb4d04c-c013-f89a-35b7-9c5d063c11e8"
      },
      "source": [
        "## events.csv"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "332d0e7c-56b0-6a06-5a96-8083105098a3"
      },
      "outputs": [],
      "source": [
        "ev = pd.read_csv('../input/events.csv', nrows=10)\n",
        "ev"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "9f004687-e741-9c38-a173-22b116e923bb"
      },
      "source": [
        "## page_views_sample.csv"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "24f048b1-42a3-ac44-b8fc-aa2d2c55f574"
      },
      "outputs": [],
      "source": [
        "pa = pd.read_csv('../input/page_views_sample.csv', nrows=10)\n",
        "pa"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "1e170231-0ae2-129b-0a99-cdf6cf967c80"
      },
      "source": [
        "## promoted_content.csv"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f9f13013-466c-fc27-ddbe-f4622d22efa8"
      },
      "outputs": [],
      "source": [
        "pr_co = pd.read_csv('../input/promoted_content.csv', nrows=10)\n",
        "pr_co"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "d530b15d-d137-85d0-67e0-558d5c5a551b"
      },
      "source": [
        "# try a simple sorting algorithm"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "c2d3c102-5b62-e79b-e464-4e0567e310da"
      },
      "source": [
        "from https://www.kaggle.com/clustifier/outbrain-click-prediction/btb-0-63523-evaluation/code"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "ae4ad485-a971-0f7d-dc7a-6fa5c0cf7386"
      },
      "outputs": [],
      "source": [
        "import pandas as pd\n",
        "import numpy as np "
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "d5d60d77-4271-f022-eec6-4bd948b019d4"
      },
      "outputs": [],
      "source": [
        "reg = 10 # trying anokas idea of regularization\n",
        "eval = True\n",
        "\n",
        "train = pd.read_csv(\"../input/clicks_train.csv\")\n",
        "\n",
        "if eval:\n",
        "\tids = train.display_id.unique()\n",
        "\tids = np.random.choice(ids, size=len(ids)//10, replace=False)\n",
        "\n",
        "\tvalid = train[train.display_id.isin(ids)]\n",
        "\ttrain = train[~train.display_id.isin(ids)]\n",
        "\t\n",
        "\tprint (valid.shape, train.shape)\n",
        "\n",
        "cnt = train[train.clicked==1].ad_id.value_counts() #total count of clicked ad_id\n",
        "cntall = train.ad_id.value_counts() # total count of all the ad_id, use to normalize"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "b331580a-90b8-5460-70bc-ecfcfae7b49b"
      },
      "outputs": [],
      "source": [
        "del train\n",
        "\n",
        "def get_prob(k):\n",
        "    if k not in cnt:\n",
        "        return 0\n",
        "    return cnt[k]/(float(cntall[k]) + reg)\n",
        "\n",
        "def srt(x):\n",
        "    ad_ids = map(int, x.split())\n",
        "    ad_ids = sorted(ad_ids, key=get_prob, reverse=True)\n",
        "    return \" \".join(map(str,ad_ids)) "
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "87e521d0-622b-1b7c-9707-7dc54759ef4d"
      },
      "outputs": [],
      "source": [
        "if eval:\n",
        "\tfrom ml_metrics import mapk\n",
        "\t\n",
        "\ty = valid[valid.clicked==1].ad_id.values\n",
        "\ty = [[_] for _ in y]\n",
        "\tp = valid.groupby('display_id').ad_id.apply(list)\n",
        "\tp = [sorted(x, key=get_prob, reverse=True) for x in p]\n",
        "\t\n",
        "\tprint (mapk(y, p, k=12))\n",
        "else:\n",
        "\tsubm = pd.read_csv(\"../input/sample_submission.csv\") \n",
        "\tsubm['ad_id'] = subm.ad_id.apply(lambda x: srt(x))\n",
        "\tsubm.to_csv(\"subm_reg_1.csv\", index=False)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "43b6f792-098d-8263-fc1b-996ad66da1ab"
      },
      "outputs": [],
      "source": [
        "subm = pd.read_csv(\"../input/sample_submission.csv\", nrows=10)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "9b227100-fc10-0e03-4930-f162a8bb5622"
      },
      "outputs": [],
      "source": [
        "subm"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "13e1f71e-2da0-8ec9-4aba-fe343b24fa9f"
      },
      "outputs": [],
      "source": [
        "subm.ad_id.apply(lambda x: srt(x))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "ab69de33-bcaa-fe28-a97e-4d96aa29cd7f"
      },
      "outputs": [],
      "source": [
        "re = subm.ad_id.apply(lambda x: [get_prob(i) for i in map(int, x.split())])"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "687b2e63-8c85-c79f-2fca-2844dabd3b34"
      },
      "outputs": [],
      "source": [
        "re[0]"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "d9479d22-c273-f6d8-264b-b3561034050f"
      },
      "outputs": [],
      "source": ""
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.5.2"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}