{"cells":[
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport sys\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier, BaggingClassifier\nfrom sklearn.metrics import accuracy_score\n"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# Taking care of data leak:\n# Match test rows from train by user_location_country, \n#                               user_location_region, \n#                               user_location_city, \n#                               hotel_market, \n#                               orig_destination_distance and get hotel_clusters for free!\n\n# 1: Create a dataset that is able to train a classifier correctly.\n#    The problem here is that even though we could train the classifier using the clicks and bookings,\n#    the test set has no click events so the information learned there could be useless and misleading.\n#    We have to take into account that using the clicks could teach us the way to recognize\n#    what the user do not want to book, but that information could be marginal when the aim is to \n#    find the hotel they actually booked.\n#    Investigate the option of formatting the train set in a way that mirrors the test set by\n#    putting all the information in one row.\n#\n# 2: Train a classifier suitable to the problem.\n#    We have to use a classifier that gives class probabilities in the end in order to find\n#    the top 5 most probable hotel_clusters.\n#\n# 3: Data leak.\n#    After fitting and evaulating the classifier(s), take care of the data leak oulined above.\n#\n# 4: DO IT AGAIN.\n#    Don't get discouraged as always. 6 weeks are a lot. You usually give up after one week and the \n#    first few unsuccessful attempts to get a better score you lazy scumbag. You want to become a \n#    full-stack data scientist, so behave like one.\n#\n# 5: Have a beer and congratulate yourself on the hard work you have done during these 46 days.\n#    Unless you gave up after 2 weeks you fucker. Then try again next time you bastard.\n\n"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# 1: Create a dataset that is able to train a classifier correctly.\n#    First instinct: Remove the clicks, keep the bookings. \n#    FROM: RandomForest_test_20160418 by Evgenii Zhukov\n\ntrain_cols = ['site_name', 'user_location_region', 'is_package', 'srch_adults_cnt', 'srch_children_cnt', 'srch_destination_id', 'hotel_market', 'hotel_country', 'hotel_cluster']\ntrain = pd.DataFrame(columns=train_cols)\ntrain_chunk = pd.read_csv('../input/train.csv', chunksize=100000)\n\nfor chunk in train_chunk:\n    train = pd.concat( [ train, chunk[chunk['is_booking']==1][train_cols] ] )\n    \ntrain.head()\ntrain_X = train[['site_name', 'user_location_region', 'is_package', 'srch_adults_cnt', 'srch_children_cnt', 'srch_destination_id', 'hotel_market', 'hotel_country']].values\ntrain_y = train['hotel_cluster'].values\n\n# 2: Train a classifier suitable to the problem.\n#    FROM: RandomForest_test_20160418 by Evgenii Zhukov\n\nrf = RandomForestClassifier(n_estimators=50, max_depth=10, random_state=2016, n_jobs=4)\nclf = BaggingClassifier(rf, n_estimators=2, max_samples=0.1, random_state=2014, n_jobs=4)\nclf.fit(train_X, train_y)\n\ntest_y = np.array([])\ntest_chunk = pd.read_csv('../input/test.csv', chunksize=50000)\n\nfor i, chunk in enumerate(test_chunk):\n    test_X = chunk[['site_name', 'user_location_region', 'is_package', 'srch_adults_cnt', 'srch_children_cnt', 'srch_destination_id', 'hotel_market', 'hotel_country']].values\n    if i > 0:\n        test_y = np.concatenate( [test_y, clf.predict_proba(test_X)])\n    else:\n        test_y = clf.predict_proba(test_X)\n    print(i)\n    \ndef get5Best(x):    \n    return \" \".join([str(int(z)) for z in x.argsort()[::-1][:5]])\n\nsubmit = pd.read_csv('../input/sample_submission.csv')\nsubmit['hotel_cluster'] = np.apply_along_axis(get5Best, 1, test_y)\nsubmit.head()\nsubmit.to_csv('Zhukov_0425.csv', index=False)\n\n# 3: Data leak.\n\n"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": ""
 }
],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}}, "nbformat": 4, "nbformat_minor": 0}