{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.5.2"}},"nbformat":4,"nbformat_minor":0,"cells":[{"metadata":{"_cell_guid":"faf45da0-997a-ae99-ca91-59149865cf0f","_active":false,"collapsed":false},"source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":2,"cell_type":"code","outputs":[],"execution_state":"idle"},{"metadata":{"_cell_guid":"45b42bf6-4b49-043d-ec79-29aae3d3ea57","_active":false,"collapsed":false},"source":"from scipy.sparse import lil_matrix\n\nnp.random.seed(2017)","execution_count":3,"cell_type":"code","outputs":[],"execution_state":"idle"},{"metadata":{"_cell_guid":"196698cd-20e2-6525-a24d-2f972049b936","_active":false,"collapsed":false},"source":"datadir_raw = '../input/'\ndtypes = {'display_id': int, 'ad_id': int, 'clicked': int}\ntrain = pd.read_csv(datadir_raw + 'clicks_train.csv', dtype=dtypes)","execution_count":5,"cell_type":"code","outputs":[],"execution_state":"idle"},{"metadata":{"_cell_guid":"06c07e57-6bbc-ead6-7287-e4b23146d517","_active":false,"collapsed":false},"source":"# assign each of the 16,874,593 display_id to one of 100 folds\ndil = train.display_id.unique()\ndfa = np.random.randint(100, size=dil.size)\ndif_dict = dict(zip(dil, dfa))\ntrain['fold'] = train.display_id.apply(dif_dict.get)","execution_count":6,"cell_type":"code","outputs":[],"execution_state":"busy"},{"metadata":{"_cell_guid":"1f816b2a-4496-48ef-31cd-e280ab4e1c94","_active":false,"collapsed":false},"source":"# using folds 0-97 for training, 98-99 for testing\nclicks_sum = train[train.fold < 98].groupby('ad_id').clicked.sum()\nclicks_count = train[train.fold < 98].groupby('ad_id').clicked.count()\n\ndef make_p_click(sm=1, tend=0.5):\n    return (clicks_sum + sm * tend) / (clicks_count + sm)","execution_count":null,"cell_type":"code","outputs":[],"execution_state":"busy"},{"metadata":{"_cell_guid":"7029e46e-4bb9-4cce-a1c2-5625feabd3aa","_active":false,"collapsed":false},"source":"def click_prob(ad):\n    '''find prob even when not all test ads will have been seen in train'''\n    try:\n        prob = p_click[ad]\n    except:\n        prob = 0\n    return prob","execution_count":null,"cell_type":"code","outputs":[],"execution_state":"busy"},{"metadata":{"_cell_guid":"1e644341-149c-2bd9-bb23-0bfa05271c0a","_active":true,"collapsed":false},"source":"def pred_score(group):\n    vals = group.values[:,[1,2]]\n    ads = vals[:,0]\n    clicks = vals[:,1]\n    cla = ads[clicks==1][0]\n    probs = [-click_prob(ad) for ad in ads];\n    cpoa = ads[np.argsort(probs)]\n    return 1 / (np.where(cpoa==cla)[0][0] + 1)","execution_count":null,"cell_type":"code","outputs":[]}]}