{"cells":[
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\n#print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output."
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "import datetime\n"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train1 = pd.read_csv(\"../input/train.csv\", parse_dates=['date_time'], nrows=1000000)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "test1 = pd.read_csv(\"../input/test.csv\")"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "#train1test1 = train1test.ix[1000:,:]"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "#train1test1.info()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "test1['date_time'] = pd.to_datetime(test1[\"date_time\"])\ntest1['year'] = test1['date_time'].dt.year\ntest1['month'] = test1['date_time'].dt.month\ntest1['day_of_week'] = test1['date_time'].dt.dayofweek\ntest1['day'] = test1['date_time'].dt.day\ntest1['hour'] = test1['date_time'].dt.hour"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train1['date_time'] = pd.to_datetime(train1[\"date_time\"])\ntrain1['year'] = train1['date_time'].dt.year\ntrain1['month'] = train1['date_time'].dt.month\ntrain1['day_of_week'] = train1['date_time'].dt.month\ntrain1['day'] = train1['date_time'].dt.day\ntrain1['hour'] = train1['date_time'].dt.hour"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train1 = train1.drop('day_of_week', axis=1)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train1['day_of_week'] = train1['date_time'].dt.dayofweek"
 },
 {
  "cell_type": "markdown",
  "metadata": {},
  "source": "train1test1['date_time'] = pd.to_datetime(train1test1[\"date_time\"])\ntrain1test1['year'] = train1test1['date_time'].dt.year\ntrain1test1['month'] = train1test1['date_time'].dt.month\ntrain1test1['day_of_week'] = train1test1['date_time'].dt.month\ntrain1test1['day'] = train1test1['date_time'].dt.day\ntrain1test1['hour'] = train1test1['date_time'].dt.hour"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train1.ix[(train1['hour'] >= 10) & (train1['hour'] < 18), 'hour'] = 1\ntrain1.ix[(train1['hour'] >= 18) & (train1['hour'] < 22), 'hour'] = 2\ntrain1.ix[(train1['hour'] >= 22) & (train1['hour'] == 24), 'hour'] = 3\ntrain1.ix[(train1['hour'] >= 1) & (train1['hour'] < 10), 'hour'] = 3"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "test1.ix[(test1['hour'] >= 10) & (test1['hour'] < 18), 'hour'] = 1\ntest1.ix[(test1['hour'] >= 18) & (test1['hour'] < 22), 'hour'] = 2\ntest1.ix[(test1['hour'] >= 22) & (test1['hour'] == 24), 'hour'] = 3\ntest1.ix[(test1['hour'] >= 1) & (test1['hour'] < 10), 'hour'] = 3"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train1 = train1.fillna(-1)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "test1 = test1.fillna(-1)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "#train1test1 = train1test1.fillna(-1)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "#train3 = train1"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train1_b = train1[train1['is_booking'] == 1].drop('is_booking', axis=1)"
 },
 {
  "cell_type": "markdown",
  "metadata": {},
  "source": "train1test1['is_booking'] = -1"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "hotelCluster = train1_b.ix[:,'hotel_cluster']"
 },
 {
  "cell_type": "markdown",
  "metadata": {},
  "source": "hotelClustertest = train1test1.ix[:,'hotel_cluster']"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train1_b = train1_b.drop('hotel_cluster', axis=1) #df.drop('reports', axis=1)"
 },
 {
  "cell_type": "markdown",
  "metadata": {},
  "source": "train1test1 = train1test1.drop('hotel_cluster',axis=1)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "test1_b.info()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train1_b.info()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train1_b = train1_b[['orig_destination_distance','srch_destination_id','srch_destination_type_id','year','month','day_of_week','hour']]"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "test1_b = test1[['orig_destination_distance','srch_destination_id','srch_destination_type_id','year','month','day_of_week','hour']]"
 },
 {
  "cell_type": "markdown",
  "metadata": {},
  "source": "train3 = train3.drop('srch_co', axis=1)"
 },
 {
  "cell_type": "markdown",
  "metadata": {},
  "source": "train1test1 = train1test1.drop('srch_co', axis=1)"
 },
 {
  "cell_type": "markdown",
  "metadata": {},
  "source": "train3 = train3.drop('date_time', axis=1)"
 },
 {
  "cell_type": "markdown",
  "metadata": {},
  "source": "train1test1 = train1test1.drop('date_time', axis=1)"
 },
 {
  "cell_type": "markdown",
  "metadata": {},
  "source": "train3.info()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "from sklearn.neural_network import MLPClassifier"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "clf = MLPClassifier(algorithm='l-bfgs', alpha=1e-5, hidden_layer_sizes=(15, 100), random_state=1)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "clf.fit(train1_b, hotelCluster) "
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "y_probb=clf.predict_proba(test1_b)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "y_probb[5130,:]"
 }
],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}}, "nbformat": 4, "nbformat_minor": 0}