{"nbformat":4,"nbformat_minor":0,"metadata":{"kernelspec":{"name":"python3","language":"python","display_name":"Python 3"},"language_info":{"pygments_lexer":"ipython3","name":"python","nbconvert_exporter":"python","file_extension":".py","codemirror_mode":{"version":3,"name":"ipython"},"mimetype":"text/x-python","version":"3.6.0"}},"cells":[{"execution_count":null,"metadata":{"_uuid":"0ada4450d405cab1398982aa02e6ffdb8a014044","_execution_state":"idle"},"cell_type":"code","outputs":[],"source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output."},{"execution_count":null,"metadata":{"_uuid":"d056f0fd63d144b4b7e78e8096d458360a05c6ca","collapsed":false,"_execution_state":"idle"},"cell_type":"code","outputs":[],"source":"train = pd.read_csv('../input/train.csv',\n                    dtype={'is_booking':bool,'srch_destination_id':np.int32, 'hotel_cluster':np.int32},\n                    usecols=['srch_destination_id','is_booking','hotel_cluster'],\n                    chunksize=1000000)\naggs = []\nprint('-'*38)\nfor chunk in train:\n    agg = chunk.groupby(['srch_destination_id',\n                         'hotel_cluster'])['is_booking'].agg(['sum','count'])\n    agg.reset_index(inplace=True)\n    aggs.append(agg)\n    print('.',end='')\nprint('')\naggs = pd.concat(aggs, axis=0)\naggs.head()"},{"execution_count":null,"metadata":{"_uuid":"4ced316204e7be1653303cafc31ad23c12349ed5","collapsed":false,"_execution_state":"idle"},"cell_type":"code","outputs":[],"source":"CLICK_WEIGHT = 0.05\nagg = aggs.groupby(['srch_destination_id','hotel_cluster']).sum().reset_index()\nagg['count'] -= agg['sum']\nagg = agg.rename(columns={'sum':'bookings','count':'clicks'})\nagg['relevance'] = agg['bookings'] + CLICK_WEIGHT * agg['clicks']\nagg.head()"},{"execution_count":null,"metadata":{"_uuid":"fe5f4e0cad19056cff08547f228066287f8f8cc7","collapsed":false,"_execution_state":"idle"},"cell_type":"code","outputs":[],"source":"def most_popular(group, n_max=5):\n    relevance = group['relevance'].values\n    hotel_cluster = group['hotel_cluster'].values\n    most_popular = hotel_cluster[np.argsort(relevance)[::-1]][:n_max]\n    return np.array_str(most_popular)[1:-1] # remove square brackets"},{"execution_count":null,"metadata":{"_uuid":"11e5359432867728e08ca74b86e7f84424d792dc","collapsed":false,"_execution_state":"idle"},"cell_type":"code","outputs":[],"source":"most_pop = agg.groupby(['srch_destination_id']).apply(most_popular)\nmost_pop = pd.DataFrame(most_pop).rename(columns={0:'hotel_cluster'})\nmost_pop.head(20)"},{"execution_count":null,"metadata":{"_uuid":"5aae3fb745a599ee037767ae7c0404ad4e31afb3","collapsed":false,"_execution_state":"idle"},"cell_type":"code","outputs":[],"source":"relevance = agg['relevance'].values"},{"execution_count":null,"metadata":{"_uuid":"c64b2cb973e20330796685050e9ab12b05836be4","collapsed":false,"_execution_state":"idle"},"cell_type":"code","outputs":[],"source":"np.aggsort(relevance)[:4]"},{"execution_count":null,"metadata":{"_uuid":"19c5286db33f11f83d7d6686ba9b147af8400ef5","collapsed":false,"_execution_state":"idle"},"cell_type":"code","outputs":[],"source":""}]}