{"cells":[{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"0af5c8bf-03d5-f3f5-d395-2dafa84d1b15"},"outputs":[],"source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output."},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"bdb8ee6f-668f-3c3f-c4c3-0a403192e66d"},"outputs":[],"source":"\n\ndf_train = pd.read_csv( \"../input/clicks_train.csv\", nrows=100000)\n"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"e06b451d-8a86-df91-4bb3-bfa42ef3f706"},"outputs":[],"source":"df_train"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"3d73a9ca-2c5d-9ea7-bd16-89055fffc01e"},"outputs":[],"source":"# ----------- \n# model training goes here ... let's borrow from the \"pandas is cool\" script as a quick example\n# ----------- \n\nad_likelihood = df_train.groupby('ad_id').clicked.agg(['count','sum' ]).reset_index()\nM             = df_train.clicked.mean()"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"1c16a16d-f6a0-969f-0d9c-613421d9d34d"},"outputs":[],"source":"ddad_likelihood"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"3252034a-cd79-00c8-cafa-590bb7bd3555"},"outputs":[],"source":"\n\nad_likelihood['likelihood'] = (ad_likelihood['sum'] + 12*M) / (12 + ad_likelihood['count'])"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"2be44610-9858-d69c-d2cb-c38c40eb9730"},"outputs":[],"source":"\n\ndf_train = df_train.merge(ad_likelihood, how='left')"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"f2fd399b-dcde-c248-9e80-3e573d2c4de1"},"outputs":[],"source":"\n\ndf_train.likelihood.fillna(M, inplace=True)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"cbc771ec-7c30-79f1-764c-e88ceec48f4f"},"outputs":[],"source":"dddf_train"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"f9df13bd-c5b4-2af8-95c5-6d8d924f5152"},"outputs":[],"source":"\n\n# ----------- \n# NOTE: this approach is specific to this particular competition\n#\n# The MAP metric here just boils down to knowing where you ended up ranking the actual ad that was clicked, relative\n# to the other ads in its display context.\n# ----------- \n\ndf_train.sort_values(['display_id', 'likelihood'], inplace=True, ascending=[True, False] )\n\n"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"3c92410a-2dbd-b4d6-0ed7-0e5aedcd4e32"},"outputs":[],"source":"df_train.head(10)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"b3878e63-7179-36f0-a5ef-e3c762d8fad0"},"outputs":[],"source":"# -------\n# Slower way\n#\n%time\nfrom ml_metrics import mapk\n\nY_ads = df_train[ df_train.clicked == 1 ].ad_id.values.reshape(-1,1)\nP_ads = df_train.groupby(by='display_id', sort=False).ad_id.apply( lambda x: x.values ).values\n\nscore = mapk( Y_ads, P_ads, 12 )\n\nprint(\"MAP: %.12f\" % score)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"8d259d34-a57f-2d95-0b72-6267ca640908"},"outputs":[],"source":"Y_ads"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"35461d4a-8394-7a68-bb53-8e009ef36b12"},"outputs":[],"source":"P_ads"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"e2f9e609-01a2-6491-dd50-8e0085f8a56e"},"outputs":[],"source":"%time\n\n# -------\n# Now this is a quicker way to evaluate your score without needing to groupby or use the default MAP@ functions.\n#\n# After sorting each context in order of decreasing predicted probability, and giving each row in the dataset a sequential \n# index, then the delta between the index of the clicked ad, and the index of the first ad in the context, will give you\n# the relative rank of the clicked ad within each context.\n\ndf_train[\"seq\"] = np.arange(df_train.shape[0])\nY_seq           = df_train[ df_train.clicked == 1 ].seq.values\nY_first         = df_train[['display_id', 'seq']].drop_duplicates(subset='display_id', keep='first').seq.values\nY_ranks         = Y_seq - Y_first\n\n# At this point, some simplification of the MAP function given what we know about this competition gives us this quick calc\n\nscore           = np.mean( 1.0 / (1.0 + Y_ranks) )\n\nprint(\"MAP: %.12f\" % score)\n"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"697a05f8-01e8-f20d-17eb-23d72344cff2"},"outputs":[],"source":""}],"metadata":{"_change_revision":0,"_is_fork":false,"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.5.2"}},"nbformat":4,"nbformat_minor":0}