{
  "metadata": {
    "kernelspec": {
      "name": "python"
    },
    "language_info": {
      "name": "python",
      "version": "3.5.1"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0,
  "cells": [
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "6e2471c8-6f2e-20b2-0fcd-85219872aabe",
        "_active": false
      },
      "outputs": [],
      "source": "# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output."
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "5f96586b-2f0c-4239-a711-2c9678079d67",
        "_active": false
      },
      "outputs": [],
      "source": "\n\nreg = 10 # trying anokas idea of regularization\neval = True\n\ntrain = pd.read_csv(\"../input/clicks_train.csv\")\n\nif eval:\n\tids = train.display_id.unique()\n\tids = np.random.choice(ids, size=len(ids)//10, replace=False)\n\n\tvalid = train[train.display_id.isin(ids)]\n\ttrain = train[~train.display_id.isin(ids)]\n\t\n\tprint (valid.shape, train.shape)\n\ncnt = train[train.clicked==1].ad_id.value_counts()\ncntall = train.ad_id.value_counts()\ndel train\n\ndef get_prob(k):\n    if k not in cnt:\n        return 0\n    return cnt[k]/(float(cntall[k]) + reg)\n\ndef srt(x):\n    ad_ids = map(int, x.split())\n    ad_ids = sorted(ad_ids, key=get_prob, reverse=True)\n    return \" \".join(map(str,ad_ids))\n   \nif eval:\n\tfrom ml_metrics import mapk\n\t\n\ty = valid[valid.clicked==1].ad_id.values\n\ty = [[_] for _ in y]\n\tp = valid.groupby('display_id').ad_id.apply(list)\n\tp = [sorted(x, key=get_prob, reverse=True) for x in p]\n\t\n\tprint (mapk(y, p, k=12))\nelse:\n\tsubm = pd.read_csv(\"../input/sample_submission.csv\") \n\tsubm['ad_id'] = subm.ad_id.apply(lambda x: srt(x))\n\tsubm.to_csv(\"subm_reg_1.csv\", index=False)"
    },
    {
      "metadata": {
        "_cell_guid": "c99019db-51d1-521f-3be0-e8ebbd064cff",
        "_active": true,
        "collapsed": false
      },
      "source": null,
      "execution_count": null,
      "cell_type": "code",
      "outputs": []
    }
  ]
}