{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.5.2"}},"nbformat":4,"nbformat_minor":0,"cells":[{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"254284d8-727a-4300-35fe-1a7cd28e14a7"},"outputs":[],"source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output."},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"f275a356-e2bd-c504-3ac2-ceeb857dfd95"},"outputs":[],"source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom scipy import sparse\n%matplotlib inline"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"b2952fba-650d-acbb-7685-7bb806c5c75a"},"outputs":[],"source":"train = pd.read_csv('../input/clicks_train.csv',usecols=['ad_id', 'clicked'])"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"093f32e4-7648-5dc0-d70d-fb7f9977a979"},"outputs":[],"source":"ad_train_likelehood = train.groupby('ad_id')['clicked'].agg(['count', 'sum', 'mean']).reset_index()"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"ad6bc7e6-c00f-5fc4-3033-eaaa8690aed6"},"outputs":[],"source":"M = train.clicked.mean()\ndel train"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"026a28d5-ad07-c8f2-0f72-487ad49562e3"},"outputs":[],"source":"M"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"69e14863-0924-2924-90f5-b3024c78e037"},"outputs":[],"source":"ad_train_likelehood['likelihood'] = (ad_train_likelehood['sum'] + 12*M) / (12 + ad_train_likelehood['count'])\n\ntest = pd.read_csv(\"../input/clicks_test.csv\")\ntest = test.merge(ad_train_likelehood, how='left')\ntest.likelihood.fillna(M, inplace=True)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"048f8d51-7ade-0749-127f-2989a58e6459"},"outputs":[],"source":"test.sort_values(['display_id','likelihood'], inplace=True, ascending=False)\nsubm = test.groupby('display_id')['ad_id'].apply(lambda x: \" \".join(map(str,x))).reset_index()\nsubm.to_csv(\"subm.csv\", index=False)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"44c65a57-2e86-64fa-165a-1e942371bfba"},"outputs":[],"source":null}]}