{
  "metadata": {
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.5.2"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0,
  "cells": [
    {
      "metadata": {
        "_cell_guid": "144e6298-a594-35b2-a555-78f8431bc57d",
        "_active": false,
        "collapsed": false
      },
      "source": "# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output.\nprint(check_output([\"ls\", \"../working\"]).decode(\"utf8\"))",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "7c512a87-f9ac-5572-758a-69972a3cb6b7",
        "_active": false,
        "collapsed": false
      },
      "source": "clicks_train = pd.read_csv(\"../input/clicks_train.csv\")\nmini_clicks_train = clicks_train.sample(40000, random_state = 0)\nmini_clicks_train.to_csv(\"mini_clicks_train.csv\")\n\n#get an error on this, \"../input/page_views.csv\" does not exist idk\n#page_views = pd.read_csv(\"../input/page_views.csv\")\n#mini_page_views = page_views[page_views[\"document_id\"].isin(mini_promoted[\"document_id\"])]\n#mini_page_views.to_csv(\"mini_page_views.csv\")\n\n",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "e1c7ec44-da94-f4b4-9276-a2eaf11eebaf",
        "_active": false,
        "collapsed": false
      },
      "source": "promoted_content = pd.read_csv(\"../input/promoted_content.csv\")\nmini_promoted = promoted_content[promoted_content[\"ad_id\"].isin(mini_clicks_train[\"ad_id\"])]\nmini_promoted.to_csv(\"mini_promoted.csv\")\n",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "cee4da70-b343-c1b9-55e4-d8bda1fb09af",
        "_active": false,
        "collapsed": false
      },
      "source": "doc_cats = pd.read_csv(\"../input/documents_categories.csv\")\nmini_doc_cats = doc_cats[doc_cats[\"document_id\"].isin(mini_promoted[\"document_id\"])]\nmini_doc_cats.to_csv(\"mini_doc_cats.csv\")\n",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "26d22bc5-9a04-f96b-8871-c0ada75de6c4",
        "_active": false,
        "collapsed": false
      },
      "source": "doc_ents = pd.read_csv(\"../input/documents_entities.csv\")\nmini_doc_ents = doc_ents[doc_ents[\"document_id\"].isin(mini_promoted[\"document_id\"])]\nmini_doc_ents.to_csv(\"mini_doc_ents.csv\")\n",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "bb69dc6e-dca4-42f3-7e77-e6313d8f3935",
        "_active": false,
        "collapsed": false
      },
      "source": "doc_meta = pd.read_csv(\"../input/documents_meta.csv\")\nmini_doc_meta = doc_meta[doc_meta[\"document_id\"].isin(mini_promoted[\"document_id\"])]\nmini_doc_meta.to_csv(\"mini_doc_meta.csv\")\n",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "645f0c14-b815-a915-250f-c95041aff8dc",
        "_active": false,
        "collapsed": false
      },
      "source": "doc_topics = pd.read_csv(\"../input/documents_topics.csv\")\nmini_doc_topics = doc_topics[doc_topics[\"document_id\"].isin(mini_promoted[\"document_id\"])]\nmini_doc_topics.to_csv(\"mini_doc_topics.csv\")\n",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "431eb6c6-a921-6da1-fd4b-85f5a4506dee",
        "_active": false,
        "collapsed": false
      },
      "source": "events = pd.read_csv(\"../input/events.csv\")\nmini_events = events[events[\"display_id\"].isin(mini_clicks_train[\"display_id\"])]\nmini_events.to_csv(\"mini_events.csv\")\n\n",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "0223c29c-9c74-bf64-7d09-a419d070c2d8",
        "_active": false,
        "collapsed": false
      },
      "source": "# Code for full datasets\n\n#clicks_train = pd.read_csv(\"clicks_train.csv\")\n#events = pd.read_csv(\"events.csv\") \n#promoted = pd.read_csv(\"promoted_content.csv\")\n\n#doc_cats = pd.read_csv(\"documents_categories.csv\")\n#doc_ents = pd.read_csv(\"documents_entities.csv\")\n#doc_meta = pd.read_csv(\"documents_meta.csv\")\n#doc_topics = pd.read_csv(\"documents_topics.csv\")",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "a7f9c637-0671-29ee-45c0-de2338db4076",
        "_active": false,
        "collapsed": false
      },
      "source": "# Mini-Data Set Preparation",
      "execution_count": null,
      "cell_type": "markdown",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "0b77718b-6160-029e-c4f0-ed79498c0e55",
        "_active": false,
        "collapsed": false
      },
      "source": "After the Kaggle Script \"Making a mini-data set\" is run (FYI, it takes about 2 minutes to run) to reduce the size of the data to 40,000 instances, run this script to organize data into a single dataframe. \n\nRun this with the 8 csv files produced by the Kaggle Script in the same directory. ",
      "execution_count": null,
      "cell_type": "markdown",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "bb9e6bb3-3798-d199-ce6e-18bad6318c0b",
        "_active": false,
        "collapsed": false
      },
      "source": "Note: This is a Python3 script because that is what Kaggle uses. ",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "f77e22cd-0282-7430-e887-bd6d99759fe5",
        "_active": false,
        "collapsed": false
      },
      "source": "clicks_train = pd.read_csv(\"mini_clicks_train.csv\")#got\ndoc_cats = pd.read_csv(\"mini_doc_cats.csv\")\ndoc_ents = pd.read_csv(\"mini_doc_ents.csv\")\ndoc_meta = pd.read_csv(\"mini_doc_meta.csv\")\ndoc_topics = pd.read_csv(\"mini_doc_topics.csv\")\nevents = pd.read_csv(\"mini_events.csv\") #got\n#page_views = pd.read_csv(\"mini_page_views.csv\") Once I get this imported\npromoted = pd.read_csv(\"mini_promoted.csv\")#got",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "0951e15d-95c3-4237-3df0-df8ba5686926",
        "_active": false,
        "collapsed": false
      },
      "source": "## Join clicks_train and events on display_id",
      "execution_count": null,
      "cell_type": "markdown",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "d228aa27-aa73-e785-321f-99824dc2db1f",
        "_active": false,
        "collapsed": false
      },
      "source": "#clicks_train and events have a 1:1 relationship\nprint(len(events[\"display_id\"].unique()))\nprint(len(clicks_train[\"display_id\"].unique()))",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "15463bca-57b0-18cf-0e13-edf8eda99da7",
        "_active": false,
        "collapsed": false
      },
      "source": "#the first column seems to be the old index, we don't need this\nclicks_train = clicks_train.set_index('display_id')\ndel clicks_train[\"Unnamed: 0\"]\nclicks_train.head()",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "c4ee73d7-2c70-21e7-1874-f142e8fe2645",
        "_active": false,
        "collapsed": false
      },
      "source": "del events[\"Unnamed: 0\"]\nevents = events.set_index(\"display_id\")\nevents.head()",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "54405a88-9a17-f629-ba33-8024d8f1b4d1",
        "_active": false,
        "collapsed": false
      },
      "source": "data = clicks_train.join(events)\ndata.head()",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "6e9e0564-a7fb-86f2-e4d5-66b883f9613d",
        "_active": false,
        "collapsed": false
      },
      "source": "## Promoted",
      "execution_count": null,
      "cell_type": "markdown",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "333c3ee0-51ec-532a-2b12-943e5707a427",
        "_active": false,
        "collapsed": false
      },
      "source": "len(promoted)",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "2201fa71-5046-de0a-8298-4293c2e8cc4f",
        "_active": false,
        "collapsed": false
      },
      "source": "#there is not a one-to-one relationship between document_id in promoted and the master data\n#This is because the same ad is being shown in different documents I think\nprint(len(promoted[\"document_id\"].unique()))\nprint(len(data[\"document_id\"].unique()))",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "12f92bfc-cf2e-16ad-a77e-b92dcd607d84",
        "_active": false,
        "collapsed": false
      },
      "source": "promoted.head()\ndel promoted[\"Unnamed: 0\"]\ndel promoted['document_id'] #I think all we want from here is the link between ad_id and campaign id\npromoted.head()",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "ec62ed4e-551f-ca5a-a3c6-457497e7378f",
        "_active": false,
        "collapsed": false
      },
      "source": "#there is a one-to-one relationship between ad_id in promoted and the master data\nprint(len(promoted[\"ad_id\"].unique())) #each add can appear more than once\nprint(len(data[\"ad_id\"].unique()))",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "fff79929-b39e-ae1c-7feb-80cb2fc1fa0c",
        "_active": false,
        "collapsed": false
      },
      "source": "## Joining Info about each ad",
      "execution_count": null,
      "cell_type": "markdown",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "4e7a07e1-192d-137e-3b09-a81c4d4b8262",
        "_active": false,
        "collapsed": false
      },
      "source": "I make a dictionary of the advertiser and campaign id for each ad_id, map that dictionary to the ad id to make the advertizer and campain columns",
      "execution_count": null,
      "cell_type": "markdown",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "f642ccdf-2c5b-4bca-cbda-e95bb8288d2f",
        "_active": false,
        "collapsed": false
      },
      "source": "data.head()",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "4b1cb203-8e80-534d-1926-ebfd5a652f74",
        "_active": false,
        "collapsed": false
      },
      "source": "print(len(data))\nprint(len(data[\"ad_id\"].unique())) #adds appear on average slightly more than twice in our minidata set",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "3912befc-805f-90db-a7bf-ed14d6e527b8",
        "_active": false,
        "collapsed": false
      },
      "source": "#make dictionaries to look up advertizer id and campaign id for each ad_id\nadvertiser_dict = dict(zip(promoted.ad_id, promoted.advertiser_id))\ncampaign_dict = dict(zip(promoted.ad_id, promoted.campaign_id))\n",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "4b64beba-d4f9-f632-afd3-05c5a6d82546",
        "_active": false,
        "collapsed": false
      },
      "source": "data[\"campaign_id\"] = data[\"ad_id\"].map(campaign_dict)\ndata[\"advertiser_id\"] = data[\"ad_id\"].map(advertiser_dict)\ndata.head()",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "0b716ee8-9766-b3dd-a4dd-5f0ced1a57f8",
        "_active": false,
        "collapsed": false
      },
      "source": "print(len(data))\nprint(len(data[\"ad_id\"].unique())) #adds appear on average slightly more than twice in our minidata set",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "80c3d064-ee38-048a-0143-59fc87c390dd",
        "_active": false,
        "collapsed": false
      },
      "source": "## Working with Page Views",
      "execution_count": null,
      "cell_type": "markdown",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "2114240e-504e-9ab9-5201-f398f8bfd6f3",
        "_active": false,
        "collapsed": false
      },
      "source": "Can't get the pageviews file to import, will work on this later",
      "execution_count": null,
      "cell_type": "markdown",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "22d81236-2922-247a-09a6-efb82dd8c1bd",
        "_active": false,
        "collapsed": false
      },
      "source": "## Importing Document Information",
      "execution_count": null,
      "cell_type": "markdown",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "89b6e51c-0b98-97d8-363e-8eee3dbbdf30",
        "_active": false,
        "collapsed": false
      },
      "source": "I'm super stuck on why all the document ids that appear in our data arent in the files with more information about each documents.",
      "execution_count": null,
      "cell_type": "markdown",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "61b5bbed-9adf-cd98-7532-b5976891dc64",
        "_active": false,
        "collapsed": false
      },
      "source": "#Why aren't there the same number of unique documents in each of these\nprint(len(data[\"document_id\"].unique()))\nprint(len(doc_cats[\"document_id\"].unique()))\nprint(len(doc_ents[\"document_id\"].unique()))\nprint(len(doc_meta[\"document_id\"].unique()))\nprint(len(doc_topics[\"document_id\"].unique()))",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "82406168-81ee-ac63-e8f5-b42e89433a3c",
        "_active": false,
        "collapsed": false
      },
      "source": "#each document has multiple possible entities, categories, topics with different confidence level. \n#maybe we should just for now keep the most likely entity, topic and category? \ndoc_ents.head()",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "c6494ae3-6cdf-e857-1e4b-727605a9c20e",
        "_active": false,
        "collapsed": false
      },
      "source": "doc_cats.head()",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "9a56ccd0-e5ef-57bc-6294-0bd172d92ed2",
        "_active": false,
        "collapsed": false
      },
      "source": "# print (clicks_train.head())\nprint (clicks_train[0:])",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "fb0777ab-a053-d834-6412-8def3dc133f1",
        "_active": false,
        "collapsed": false
      },
      "source": "## Code to create entry",
      "execution_count": null,
      "cell_type": "markdown",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "1125dfee-a80b-71a0-7eff-29cd3c26e194",
        "_active": false,
        "collapsed": false
      },
      "source": "\nreg = 10 # trying anokas idea of regularization\neval = False # True = split off 10% of training data for validation and test performance\n\ntrain = clicks_train\n\nif eval:\n    ids = train.index.values\n    ids = np.random.choice(ids, size=len(ids)//10, replace=False)\n    notids = set(train.index.values) - set(ids)\n    \n    #print(ids)\n    #print(len(ids))\n    valid = train.loc[ids] # random 10% for validation data \n    #print(\"length of train:\" + str(len(train)))\n    #print(\"length of valid: \" + str(len(valid)))\n    #print(valid.head())\n    \n    train = train.loc[notids] # remaining 90% as training data\n    #print(train.head())\n    print (valid.shape, train.shape)\n\ncnt = train[train.clicked==1].ad_id.value_counts() # group # of clicks by ad \ncntall = train.ad_id.value_counts() # group # of displays by ad\ndel train\n\ndef get_prob(k):\n    if k not in cnt:\n        return 0\n    return cnt[k]/(float(cntall[k]) + reg)  # return the proportion of ad clicks / displays\n\ndef srt(x):\n    ad_ids = map(int, x.split()) # take in list of ads shown to each user\n    ad_ids = sorted(ad_ids, key=get_prob, reverse=True) # re-sort the ads by training clicks / displays\n    return \" \".join(map(str,ad_ids)) # return the list with the ads sorted for submission\n   \nif eval:\n    from ml_metrics import mapk\n\n    y = valid[valid.clicked==1].ad_id.values # create list of ad click counts in validation set\n    y = [[_] for _ in y]\n    p = valid.groupby('display_id').ad_id.apply(list) #TODO: Blows up because display_id is an index field\n    p = [sorted(x, key=get_prob, reverse=True) for x in p] # create list in order expected\n\n    print (mapk(y, p, k=12)) # compare predicted order vs. actual order in validation set\n\nelse:\n    subm = pd.read_csv(\"../input/sample_submission.csv\") # load the sample submission file\n    subm['ad_id'] = subm.ad_id.apply(lambda x: srt(x)) # re-sort the ads by overall training clicks / display\n    subm.to_csv(\"subm_reg_2.csv\", index=False) ",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    },
    {
      "metadata": {
        "_cell_guid": "c5330f73-f078-d09a-586a-0b54c2b9ddda",
        "_active": true,
        "collapsed": false
      },
      "source": "print(check_output([\"ls\", \"../working\"]).decode(\"utf8\"))",
      "execution_count": null,
      "cell_type": "code",
      "outputs": [],
      "execution_state": "idle"
    }
  ]
}