{"cells":[{"cell_type":"code","execution_count":null,"metadata":{"collapsed":false},"outputs":[],"source":"from __future__ import division\nimport string\nimport numpy as np\nfrom numpy.random import randn\nfrom pandas import Series, DataFrame\nimport pandas as pd\nimport csv\nimport os\nimport matplotlib\nimport matplotlib.pyplot as plt\nfrom scipy import stats\nimport time\nimport random\nimport datetime\nfrom sklearn.ensemble import GradientBoostingClassifier\n#%matplotlib inline\n\n        \n# file = open(\"../input/train.csv\")\n# fout = open('subset_datatrain.csv','w')\n# n = 0\n# fout.write(file.readline())\n# for line in file:\n#     arr = line.strip().split(',')\n#     is_book = int(arr[-6])\n#     if is_book == 1:\n#         fout.write(line)\n# fout.close()\n# file.close()\n\n\n\n# dategroup = ['2013Jan','2013May','2013Sep','2014Jan','2014May','2014Sep','2015Jan','2015May','2015Sep','2016Jan','2016May','2016Sep']\n# chingroup = ['2013Jan','2013May','2013Sep','2014Jan','2014May','2014Sep','2015Jan','2015May','2015Sep','2016Jan','2016May','2016Sep']\n# dateix = [[] for i in range(12)]\n# chinix = [[] for i in range(12)]\n\ndef datedeal(date):\n    n = len(date)\n    for i in range(n):\n        a = date[i]\n        if type(a) == type(0.1):\n            a = '2015-01-01'\n        if int(a[1])>0 or int(a[2])>1 or int(a[2])<1 or int(a[3])>6 or int(a[3])<3:   #大于2016或小于2013的年份全部换成2015\n            date[i] = '2015-01-01'\n    return pd.to_datetime(date)\n\ndef frameDateDeal(frame, datename):\n    frame[datename]=frame[datename].fillna('2015-01-01')\n    dateix = [[] for i in range(12)]\n    datevalue = datedeal(frame[datename].values)\n    datevalue = Series(np.arange(len(datevalue)),index = datevalue)\n    for i in range(48):\n        y = divmod(i,12)[0]\n        r = divmod(i,12)[1]\n        n = divmod(i,4)[0]\n        if r<9:\n            dateix[n].extend(datevalue['201'+str(3+y)+'-0'+str(r+1)].values)\n        else:\n            dateix[n].extend(datevalue['201'+str(3+y)+'-'+str(r+1)].values)\n    for i in range(12):\n        frame[datename].values[dateix[i]] = i\n    return frame\n\n# coding: utf-8\n\nimport datetime\nfrom heapq import nlargest\nfrom operator import itemgetter\nfrom collections import defaultdict\n\ndef run_solution():\n    print('Preparing arrays...')\n    f = open(\"../input/train.csv\", \"r\")\n    f.readline()\n    best_hotels_od_ulc = defaultdict(lambda: defaultdict(int))\n    best_hotels_search_dest = defaultdict(lambda: defaultdict(int))\n    best_hotels_search_dest1 = defaultdict(lambda: defaultdict(int))\n    best_hotel_country = defaultdict(lambda: defaultdict(int))\n    popular_hotel_cluster = defaultdict(int)\n    total = 0\n\n    # Calc counts\n    while 1:\n        line = f.readline().strip()\n        total += 1\n\n        if total % 10000000 == 0:\n            print('Read {} lines...'.format(total))\n\n        if line == '':\n            break\n\n        arr = line.split(\",\")\n        book_year = int(arr[0][:4])\n        user_location_city = arr[5]\n        orig_destination_distance = arr[6]\n        srch_destination_id = arr[16]\n        is_booking = int(arr[18])\n        hotel_country = arr[21]\n        hotel_market = arr[22]\n        hotel_cluster = arr[23]\n\n        append_1 = 3 + 17*is_booking\n        append_2 = 1 + 5*is_booking\n\n        if user_location_city != '' and orig_destination_distance != '':\n            best_hotels_od_ulc[(user_location_city, orig_destination_distance)][hotel_cluster] += 1\n\n        if srch_destination_id != '' and hotel_country != '' and hotel_market != '' and book_year == 2014:\n            best_hotels_search_dest[(srch_destination_id, hotel_country, hotel_market)][hotel_cluster] += append_1\n        \n        if srch_destination_id != '':\n            best_hotels_search_dest1[srch_destination_id][hotel_cluster] += append_1\n        \n        if hotel_country != '':\n            best_hotel_country[hotel_country][hotel_cluster] += append_2\n        \n        popular_hotel_cluster[hotel_cluster] += 1\n    \n    f.close()\n\n    print('Generate submission...')\n    now = datetime.datetime.now()\n    #path = 'submission_' + str(now.strftime(\"%Y-%m-%d-%H-%M\")) + '.csv'\n    path = 'predict_test.csv'\n    out = open(path, \"w\")\n    f = open(\"../input/test.csv\", \"r\")\n    f.readline()\n    total = 0\n    #out.write(\"id,hotel_cluster\\n\")\n    topclasters = nlargest(10, sorted(popular_hotel_cluster.items()), key=itemgetter(1))\n\n    while 1:\n        line = f.readline().strip()\n        total += 1\n\n        if total % 1000000 == 0:\n            print('Write {} lines...'.format(total))\n\n        if line == '':\n            break\n\n        arr = line.split(\",\")\n        id = arr[0]\n        user_location_city = arr[6]\n        orig_destination_distance = arr[7]\n        srch_destination_id = arr[17]\n        hotel_country = arr[20]\n        hotel_market = arr[21]\n        # id = arr[0]\n        # user_location_city = arr[5]\n        # orig_destination_distance = arr[6]\n        # srch_destination_id = arr[16]\n        # hotel_country = arr[21]\n        # hotel_market = arr[22]\n\n\n        out.write(str(id) + ',')\n        filled = []\n\n        out.write(str(id) + ',')\n        filled = []\n\n        s1 = (user_location_city, orig_destination_distance)\n        if s1 in best_hotels_od_ulc:\n            d = best_hotels_od_ulc[s1]\n            topitems = nlargest(10, sorted(d.items()), key=itemgetter(1))\n            for i in range(len(topitems)):\n                if topitems[i][0] in filled:\n                    continue\n                if len(filled) == 10:\n                    break\n                out.write(',' + topitems[i][0])\n                filled.append(topitems[i][0])\n\n        s2 = (srch_destination_id, hotel_country, hotel_market)\n        if s2 in best_hotels_search_dest:\n            d = best_hotels_search_dest[s2]\n            topitems = nlargest(10, d.items(), key=itemgetter(1))\n            for i in range(len(topitems)):\n                if topitems[i][0] in filled:\n                    continue\n                if len(filled) == 10:\n                    break\n                out.write(',' + topitems[i][0])\n                filled.append(topitems[i][0])\n        elif srch_destination_id in best_hotels_search_dest1:\n            d = best_hotels_search_dest1[srch_destination_id]\n            topitems = nlargest(10, d.items(), key=itemgetter(1))\n            for i in range(len(topitems)):\n                if topitems[i][0] in filled:\n                    continue\n                if len(filled) == 10:\n                    break\n                out.write(',' + topitems[i][0])\n                filled.append(topitems[i][0])\n\n        if hotel_country in best_hotel_country:\n            d = best_hotel_country[hotel_country]\n            topitems = nlargest(10, d.items(), key=itemgetter(1))\n            for i in range(len(topitems)):\n                if topitems[i][0] in filled:\n                    continue\n                if len(filled) == 10:\n                    break\n                out.write(',' + topitems[i][0])\n                filled.append(topitems[i][0])\n\n        for i in range(len(topclasters)):\n            if topclasters[i][0] in filled:\n                continue\n            if len(filled) == 10:\n                break\n            out.write(',' + topclasters[i][0])\n            filled.append(topclasters[i][0])\n\n        out.write(\"\\n\")\n    out.close()\n    print('Completed!')\n\n\ndef bicluster(i,j):\n    tix = np.array(trainpart['hotel_cluster'].values==i)+np.array(trainpart['hotel_cluster'].values==j)\n    tGBtraintarget = (trainpart['hotel_cluster'].values==i)*1\n    tGBpara = {'data':GBdata,'feature_names':featurelist,'target':tGBtraintarget,\n    'target_names':np.arange(100)}\n    tmp = tGBpara['target'][tix]\n    if sum(tmp==0)==0:\n        tmp[-1] = 0\n    tclf = GradientBoostingClassifier(n_estimators=20, learning_rate=1,\n    max_depth=4, random_state=0).fit(GBdata[tix], tmp)\n    return tclf\n\ndef oneclus(n):\n    a = list(range(100))\n    a.remove(n)\n    return bicluster(n,random.sample( a ,1)[0])\n    \ndef getvoter():\n    voterlist = []\n    for i in range(100):\n        accuracy = []\n        clflist = []\n        clf  = oneclus(i)\n        clflist.append(clf)\n        for j in range(100):\n            tix = np.array(testpart1['hotel_cluster'].values==i)+np.array(testpart1['hotel_cluster'].values==j)\n            accuracy.append( clf.score(testdata1[tix], 1*(testpart1['hotel_cluster'][tix].values==i)) )  \n            #must use a testdata that contains true clusters\n        accuracy = DataFrame([accuracy],index = ['accuracy']).T\n        clusix = accuracy.sort_values( by ='accuracy',ascending = True).index[:2]\n        tclf = clf\n        for ind in clusix:\n            tclf = bicluster(i,ind)    \n            clflist.append(tclf)\n        voterlist.append(clflist)\n    return voterlist\n \n\n\nfile = open(\"../input/train.csv\")\nfout = open('subset_datatest.csv','w')\nn = 0\nfor line in file:\n    if n == 0:\n        fout.write(line)\n    if n <200000*5:\n        n +=1\n    elif 200000*5<=n <200000*10:\n        n +=1\n        fout.write(line)\n    else:\n        break\nfout.close()\nfile.close()\nfile = open(\"../input/train.csv\")\nfout = open('subset_datatrain.csv','w')\nn = 0\nfor line in file:\n    if n <200000*5:\n        n +=1\n        fout.write(line)\n    else:\n        break\nfout.close()\nfile.close()\n\n\n\nfeaturelist = ['user_id','user_location_city','srch_destination_id','hotel_market','srch_ci']\nwhlist = ['user_id','user_location_city','srch_destination_id','hotel_market','srch_ci','hotel_cluster']\ntrainpart = pd.read_csv('subset_datatrain.csv',na_values=['--  '],usecols = whlist)\n\n#to be used in function getvoter()\ntrainpart = frameDateDeal(trainpart,'srch_ci')\nLGBdata = trainpart[featurelist].values\nLGBpara = {'data':LGBdata,'feature_names':featurelist,'target':trainpart['hotel_cluster'].values,\n'target_names':np.arange(100)}\n\n#to be used in final selection\ntestpart = pd.read_csv('../input/test.csv',na_values=['--  '],usecols = featurelist)\ntestpart = frameDateDeal(testpart,'srch_ci')\ntestdata = testpart[featurelist].values\n\n#to be used in funcion getvoter()\ntestpart1 = pd.read_csv('subset_datatest.csv',na_values=['--  '],usecols = whlist)\ntestpart1 = frameDateDeal(testpart1,'srch_ci')\ntestdata1 = testpart1[featurelist].values\n\nos.remove('subset_datatest.csv')\n\nrun_solution()\n"}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}},"nbformat":4,"nbformat_minor":0}