{"cells":[{"metadata":{"_uuid":"66bc0e57ea17d94b350d15e92f2b6c87cbaf4745"},"cell_type":"markdown","source":"**New York Taxi **\n![](http://www.nyc.gov/html/tlc/images/features/fi_industry_yellow_taxi_photo.jpg)"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"scrolled":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sb\nimport matplotlib.pyplot as plt\n%matplotlib inline\nsb.set()\n\nimport folium\n\nfrom sklearn import metrics\nfrom sklearn import metrics\nfrom scipy.stats import zscore\nimport lightgbm as lgbm\n\nimport tensorflow as tf\nfrom keras.models import Sequential\nfrom keras.layers.core import Dense, Activation\nfrom keras.callbacks import EarlyStopping\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0243d778f4eb0fbc178acf89f3b21d9509d3d715"},"cell_type":"code","source":"# Helping functions from Jeff Heaton's course on Deep Learning\n# https://github.com/jeffheaton/t81_558_deep_learning/blob/master/t81_558_class06_backpropagation.ipynb\n    \n# Convert all missing values in the specified column to the median\ndef missing_median(df, name):\n    med = df[name].median()\n    df[name] = df[name].fillna(med)\n    \ndef remove_outliers(df, name, sd):\n    drop_rows = df.index[(np.abs(df[name] - df[name].mean()) >= (sd * df[name].std()))]\n    df.drop(drop_rows, axis=0, inplace=True)\n    \n# Encode a numeric column as zscores\ndef encode_numeric_zscore(df, name, mean=None, sd=None):\n    if mean is None:\n        mean = df[name].mean()\n\n    if sd is None:\n        sd = df[name].std()\n\n    df[name] = (df[name] - mean) / sd\n    \ndef encode_text_dummy(df, name):\n    dummies = pd.get_dummies(df[name])\n    for x in dummies.columns:\n        dummy_name = \"{}-{}\".format(name, x)\n        df[dummy_name] = dummies[x]\n    df.drop(name, axis=1, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0c9ec5d5bb82dca7a5261da27d2c2a17ee3fabd9","scrolled":true},"cell_type":"code","source":"# Date parser is used to treat the date and time column data as dates\ndata_train = pd.read_csv('../input/train.csv', nrows = 10_000_000, parse_dates=['pickup_datetime'])\ndata_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c4d1624f4e0c8b6babcb7f8362d271ef3cd50cbb"},"cell_type":"code","source":"data_train.describe()\n\n# Note that there are negative fares and very large amount of fares too. \n# Longitude and Latitude are negative and also very large which is incorrrect. ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f2b56cb114f079cc511037e04ec5e136d91d79ae"},"cell_type":"code","source":"data_train.dtypes","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8b55552ff4cd5562ce5815bf7686eb64670ec93a"},"cell_type":"code","source":"data_train['fare_amount'].plot(kind='hist')\nplt.xlabel('Fare Amount')\nplt.title('Variation of Taxi Fare Amount in New York')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2ff4f5eaa19193e9e4b8c31ca5a4f98dbb3aef68"},"cell_type":"markdown","source":"****Data Cleanup****\n\nDuring this step, the data is cleaned by removing the outliers such as \n* Data with large fares,\n* Data with negative fares and negligible fares below 50 cents\n* Data with same starting to ending locations\n* Any null/na fields\n* Data with pickup/drop off beyond NY city limits\n* Data with negative person count"},{"metadata":{"trusted":true,"_uuid":"a2ff75c2d94ae4903d870471fb13b49cfe3da67a","scrolled":true},"cell_type":"code","source":"# Checking number of rides with fare over $100\ndata_train_filtered = data_train[data_train['fare_amount'] >100]\npercent_data_train_filtered = data_train_filtered.shape[0]/data_train.shape[0]*100\nprint('Number of fares greater than $100 are,',data_train_filtered.shape[0])\nprint('Percentage of fares greater than 100 ', percent_data_train_filtered)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c519b01a706678c67810def5a8f658e19b6d8322"},"cell_type":"markdown","source":"In the next step, fares greater than \\$100 as well as smaller than \\$2.50 are dropped as being outliers. Please note that $2.50 is minimum fare. "},{"metadata":{"trusted":true,"_uuid":"f482ce5c0057bbe33b0c61b0c010d634b771c548","scrolled":false},"cell_type":"code","source":"Nrows = data_train.shape[0]\nprint ('Number of rows in training data before cleaning are, {}'.format(Nrows))\n\n# Remove the outliers of pickup longitudes different than thse values which are bounding box of New York City\n# North Latitude: 40.917577 South Latitude: 40.477399 East Longitude: -73.700272 West Longitude: -74.259090\n# source https://www.mapdevelopers.com/geocode_bounding_box.php\ndata_train = data_train[data_train['dropoff_latitude'] <=40.917577]\ndata_train = data_train[data_train['dropoff_latitude'] >=40.477399]\n           \ndata_train = data_train[data_train['pickup_latitude'] <=40.917577]\ndata_train = data_train[data_train['pickup_latitude'] >=40.477399]\n\ndata_train = data_train[data_train['dropoff_longitude'] <=-73.700272]\ndata_train = data_train[data_train['dropoff_longitude'] >=-74.259090]\n           \ndata_train = data_train[data_train['pickup_longitude'] <=-73.700272]\ndata_train = data_train[data_train['pickup_longitude'] >=-74.259090]\n\n# Keep rows only with different starting and ending location\ndata_train = data_train[data_train['dropoff_latitude'] != data_train['pickup_latitude']]\ndata_train = data_train[data_train['dropoff_longitude'] != data_train['pickup_longitude']]\n\n# Replace missing rows with no fare amount with median value\nmissing_median(data_train,'fare_amount')\n\n# Removing the outliers of fare amount more than standard deviation of 3.\n#remove_outliers(data_train,'fare_amount', sd=3)\n\n# Dropping fares greater than $100\ndata_train = data_train[data_train['fare_amount']<100]\n\n# Keep rows only with non-negative fare and non-negligible erronous amount\ndata_train = data_train[data_train['fare_amount']>=2.5]\n\n# Keep rows only with non-negative fare and non-negligible erronous amount\ndata_train = data_train[data_train['passenger_count']>0]\ndata_train = data_train[data_train['passenger_count']<7]\n\nprint(\"Removing outliers with fare more than $100, negative fares, outside of boundary of NY city,negative passengers and missing na values\")\nNrows_clean = data_train.shape[0]\nPercLost = (Nrows-Nrows_clean)/Nrows*100\nprint ('Number of rows in data after cleaning are, {}'.format(Nrows_clean))\nprint ('Percentage of data removed after cleaning {}.format',PercLost)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2ebc8a51c816bbc476b8d5b4ac599efc287d24bc"},"cell_type":"code","source":"data_train.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bf40b4ac46c5a1167fc0494140ad8e709b1544db"},"cell_type":"code","source":"# Count if any data is null or zero\nprint(data_train.isnull().sum())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"275ebfccfa5af2f311e8a04b6264c93b0dab4588","scrolled":true},"cell_type":"code","source":"# Drop the data with null longitude\ndata_train = data_train[~data_train.dropoff_longitude.isnull()]\nprint(data_train.isnull().sum())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"acdca2fbf0ed10a50e931078f1aa4f752be46010","scrolled":true},"cell_type":"code","source":"import folium\nNY_CORD = (data_train['pickup_latitude'].mean(),data_train['pickup_longitude'].mean())\nmax_records = 1000\n\nny_map = folium.Map(location=NY_CORD, zoom_start=12.0,tiles= \"OpenStreetMap\")\nfor each in data_train[:max_records].iterrows():\n    folium.CircleMarker(\n        radius = 4,\n        color ='red',\n        location = [each[1]['pickup_latitude'],each[1]['pickup_longitude']],\n        popup='lat='+str(each[1]['pickup_latitude'])+',long='+str(each[1]['pickup_longitude'])\n    ).add_to(ny_map)\nny_map","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8d7ea5af0f042545df491dbd7f2ad561449a9938","scrolled":false},"cell_type":"code","source":"import folium\nNY_CORD = (data_train['pickup_latitude'].mean(),data_train['pickup_longitude'].mean())\nmax_records = 1000\n\nny_map2 = folium.Map(location=NY_CORD, zoom_start=11.0,tiles= \"OpenStreetMap\")\nfor each in data_train[:max_records].iterrows():\n    folium.CircleMarker(\n        radius = 4,\n        color ='blue',\n        location = [each[1]['dropoff_latitude'],each[1]['dropoff_longitude']],\n        popup='lat='+str(each[1]['dropoff_latitude'])+',long='+str(each[1]['dropoff_longitude'])\n    ).add_to(ny_map2)\nny_map2","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"871681561a7e4d75a54ff2e285cb6258f6f4adc7"},"cell_type":"markdown","source":"**Feature Engineering**\n\nIn this section, new features are created from locations of pickup/dropoff and time data. They allow us to look more closely into effect of these new variables on fare. \n\nIt is also necessary to consider the fare methods used by New York Taxi and Limosine Commission. \n\n* As per which, there is a surcharge of \\$1 for taxi pickup between peak hours, Monday - Friday after 4:00 PM & before 8:00 PM. \n* In addition, there is a night surcharge of \\$0.50 after 8:00 PM & before 6:00 AM. This means we need to create a feature for this. \n* There is also a flat fare plus toll charges plus surcharges for pickups to and from between JFK and Manhatten.  \n* There is a also a fixed surcharge of $17.5 for dropoff to Newark airport. \n\nUsing the pickup and dropoff locations, distance can be Haversine distance can be calculated. \nSimilarly date-time time is split into the \n* Year - to capture year over year increase in the taxi fare, \n* Day of the year -  to capture the effect of seasonal weather changes on taxi fares,\n* Day of the week - to capture effect of weekly patterns on taxi fare, such as work week Vs weekend\n* Time of the day - to capture effect of daily work schedules on taxi fares"},{"metadata":{"trusted":true,"_uuid":"710f8a28a5f228943991045234768bd923af86f6"},"cell_type":"code","source":"# Calculate the distance between two latitude and longitude data points using Haversine method\n# https://en.wikipedia.org/wiki/Haversine_formula\n# https://stackoverflow.com/questions/27928/calculate-distance-between-two-latitude-longitude-points-haversine-formula\n\n# Bearing calculated using method - https://www.movable-type.co.uk/scripts/latlong.html \n# http://mathforum.org/library/drmath/view/55417.html\n    \nimport math\n    \ndef getDistance(df):\n    R_Earth = 6371 # Radius of earth in km\n    LAT1 = df['pickup_latitude'] \n    LONG1 = df['pickup_longitude']\n    LAT2 = df['dropoff_latitude']\n    LONG2 = df['dropoff_longitude']\n    dLat = deg2rad(LAT2 - LAT1)\n    dLon = deg2rad(LONG2 - LONG1)\n    a = math.sin(dLat/2) * math.sin(dLat/2) + math.cos(deg2rad(LAT1)) * math.cos(deg2rad(LAT2)) * math.sin(dLon/2) * math.sin(dLon/2)\n    c = 2 * np.arcsin(np.sqrt(a)); \n    return R_Earth*c\n\ndef getBearing(df):\n    R_Earth = 6371 # Radius of earth in km\n    LAT1 = deg2rad( df['pickup_latitude'] )\n    LONG1 = deg2rad(  df['pickup_longitude'] )\n    LAT2 = deg2rad( df['dropoff_latitude']  )\n    LONG2 = deg2rad( df['dropoff_longitude'] )\n    dLat = (LAT2 - LAT1)\n    dLon = (LONG2 - LONG1)\n    a = np.arctan2(np.sin(dLon * np.cos(LAT2)),np.cos(LAT1) * np.sin(LAT2) - np.sin(LAT1) * np.cos(LAT2) * np.cos(dLon))\n    return a\n\ndef deg2rad(deg):\n    return np.radians(deg)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8b421b99a724838552942da8217bc3f8835eff40","scrolled":true},"cell_type":"code","source":"data_train.drop('key',1, inplace=True)\n\ndata_train['distance_km'] = data_train.apply(getDistance, axis=1)\ndata_train['bearing'] = data_train.apply(getBearing, axis=1)\n\ndef split_time(df):\n    data_train['year'] = data_train.pickup_datetime.dt.year\n    data_train['month'] = data_train.pickup_datetime.dt.month\n    data_train['day_of_year'] = data_train.pickup_datetime.dt.dayofyear\n    data_train['hour'] = data_train.pickup_datetime.dt.hour\n    data_train['day_of_week'] = data_train.pickup_datetime.dt.dayofweek\n    \nsplit_time(data_train)\n\ndata_train.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2a50cb2abd0f7a67e8c51f12c5a41a3dd01b039a"},"cell_type":"markdown","source":"To consider the peak hour and late night fares new features are created. "},{"metadata":{"trusted":true,"_uuid":"c6374524e20e02a1bed22fd9778c11784bc343b5","scrolled":false},"cell_type":"code","source":"def peak_hour(df):\n    if df['hour']>=16 and df['hour']<=20:\n        return 1\n    else:\n        return 0\n\ndef night_hour(df):\n    if df['hour']>=20 or df['hour']<=6:\n        return 1\n    else:\n        return 0\n\ndef JFK_Manhatten_pickup_drop(df):\n    # Range of co-ordinates for JFK airport which have fixed surcharge\n    if (40.6214 <= df['pickup_latitude'] <= 40.6693) and (-73.8250 <= df['pickup_longitude'] <= -73.7459) \\\n        and (40.7002 <= df['dropoff_latitude'] <= 40.7655) and (-74.0217 <= df['dropoff_longitude'] <= -73.9495) and df['fare_amount']>51:\n            return 1\n    elif (40.7002 <= df['pickup_latitude'] <= 40.7655) and (-74.0217 <= df['pickup_longitude'] <= -73.9495) \\\n        and (40.6214 <= df['dropoff_latitude'] <= 40.6693) and (-73.8250 <= df['dropoff_longitude'] <= -73.7459) and df['fare_amount']>51:\n            return 1\n    elif (40.6214 <= df['pickup_latitude'] <= 40.6693) and (-73.8250 <= df['pickup_longitude'] <= -73.7459) \\\n        and (40.7663 <= df['dropoff_latitude'] <= 40.8316) and (-73.9908 <= df['dropoff_longitude'] <= -73.9185) and df['fare_amount']>51: \n            return 1\n    elif (40.7663 <= df['pickup_latitude'] <= 40.8316) and (-73.9908 <= df['pickup_longitude'] <= -73.9185) \\\n        and (40.6214 <= df['dropoff_latitude'] <= 40.6693) and (-73.8250 <= df['dropoff_longitude'] <= -73.7459) and df['fare_amount']>51:\n            return 1\n    else:\n        return 0\n    \n\ndef NEW_airport_dropoff(df):\n    # Range of co-ordinates for Newark airport which have fixed surcharge\n    if ( (40.6694 <= df['dropoff_latitude'] <= 40.7049) and  (-74.1935 <= df['dropoff_longitude'] <= -74.1508)):\n        return 1\n    else:\n        return 0\n\ndata_train['peak'] = data_train.apply(peak_hour,axis=1)\ndata_train['night_hour'] = data_train.apply(night_hour,axis=1)\ndata_train['JFK_airport'] = data_train.apply(JFK_Manhatten_pickup_drop, axis=1)\ndata_train['NEW_airport'] = data_train.apply(NEW_airport_dropoff, axis=1)\n\n\ndata_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ecc6dff77fc5688ebd17b68652e3f4a7e70cc3ec","scrolled":true},"cell_type":"code","source":"data_train.describe()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d33bc47518757861707cdf277c8026c244f1cb59"},"cell_type":"markdown","source":"**Exploring The Data**\n\nIn this section, effect of various variables on fare amount is analyzed. "},{"metadata":{"trusted":true,"_uuid":"b98f58dc94bd2f066f11cb6c8501184c16d4a523"},"cell_type":"code","source":"data_train['fare_amount'].plot(kind='hist', bins=50)\nplt.xlabel('Fare Amount ($)')\nplt.title('Variation of Taxi Fare Amount in New York')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"85d04e8a841af985e7b1d790dae12f3eed76a9dd"},"cell_type":"code","source":"data_train['distance_km'].plot(kind='hist', bins=50)\nplt.xlabel('Distance (km)')\nplt.title('Variation of Ride Distance for a taxi in New York')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6ff6f96b8fadfe1529d5b8765bd9b70cefd2bb99"},"cell_type":"code","source":"plt.scatter(data_train['distance_km'],data_train['fare_amount'],alpha=0.2)\nplt.xlabel('Distance (km)')\nplt.ylabel('Fare ($)')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ceb7c7641bf41f49e03726be0ea6f29eae835a82"},"cell_type":"code","source":"# Plotting airport rides in a map to confirm location of pickup\ndata_train_airport = data_train[data_train['JFK_airport'] == 1]\n\n#data_train_airport.head(20)\n(data_train_airport.describe())\n# Note that mean fare is around the minimum fare for the ride is 2.5 though minimum is $52.00\n# Mean value is correct, so we have drop the values which are below $52","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7569596fd591cfe113db370ed46aed3eab1e34e9"},"cell_type":"code","source":"import folium\nJFK_CORD = (data_train_airport['pickup_latitude'].mean(),data_train_airport['pickup_longitude'].mean())\nmax_records = 1000\n\nairport_map = folium.Map(location=JFK_CORD, zoom_start=12.0,tiles= \"OpenStreetMap\")\n\nfor each in data_train_airport[:max_records].iterrows():\n    folium.CircleMarker(\n        radius = 4,\n        color ='red',\n        location = [each[1]['pickup_latitude'],each[1]['pickup_longitude']],\n        popup='lat='+str(each[1]['pickup_latitude'])+',long='+str(each[1]['pickup_longitude'])\n    ).add_to(airport_map)\nairport_map\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f32952a4db35440dfb5bcdd7166e9a42faba5fbe"},"cell_type":"code","source":"data_train_airport['fare_amount'].plot(kind='hist', bins=50)\nplt.xlabel('Fare Amount ($)')\nplt.title('Taxi Fare for rides between Manhatten and JFK')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"089effa2f64845a65cbe8bd042bf85e39fd6e1b2"},"cell_type":"code","source":"colors = np.where(data_train[\"JFK_airport\"]==1,'r','b')\ndata_train.plot.scatter('distance_km', 'fare_amount',c=colors,alpha=0.3)\nplt.xlabel('Distance (km)')\nplt.ylabel('Fare ($)')\nplt.title('Distance and Fare for the ride colored in red for airport rides')\nplt.Figure(figsize=(20, 20))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7cd84fb37f402456d60c4bfa671e1c23369363a7"},"cell_type":"code","source":"# Newark Airport Dropoff\n# Plotting airport rides in a map to confirm location of pickup\ndata_train_nw_airport = data_train[data_train['NEW_airport'] == 1]\n\n#data_train_airport.head(20)\n(data_train_nw_airport.describe())\n# Note that mean fare is around the minimum fare for the ride is 2.5 though minimum is $52.00\n# Mean value is correct, so we have drop the values which are below $52","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1bdaeda3170a464fbb5976580857ef5d1f98895e"},"cell_type":"code","source":"import folium\nNEW_CORD = (data_train_nw_airport['dropoff_latitude'].mean(),data_train_nw_airport['dropoff_longitude'].mean())\nmax_records = 1000\n\nairport_map = folium.Map(location=NEW_CORD, zoom_start=12.0,tiles= \"OpenStreetMap\")\n\nfor each in data_train_nw_airport[:max_records].iterrows():\n    folium.CircleMarker(\n        radius = 4,\n        color ='red',\n        location = [each[1]['dropoff_latitude'],each[1]['dropoff_longitude']],\n        popup='lat='+str(each[1]['dropoff_latitude'])+',long='+str(each[1]['dropoff_longitude'])\n    ).add_to(airport_map)\nairport_map","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"33034ea5134e39d36734443f6345fc0d7df9da34"},"cell_type":"code","source":"# Reference for this analysis https://www.kaggle.com/nicapotato/taxi-rides-time-analysis-and-oof-lgbm\ndef plot_time_history(df, timeframes, value, color=\"purple\"):\n    \"\"\"\n    Function to count observation occurrence through different lenses of time.\n    \"\"\"\n    f, ax = plt.subplots(len(timeframes), figsize = [12,12])\n    for i,x in enumerate(timeframes):\n        df.loc[:,[x,value]].groupby([x]).mean().plot(ax=ax[i],color=color)\n        ax[i].set_ylabel(value.replace(\"_\", \" \").title())\n        ax[i].set_title(\"{} by {}\".format(value.replace(\"_\", \" \").title(), x.replace(\"_\", \" \").title()))\n        ax[i].set_xlabel(\"\")\n    ax[len(timeframes)-1].set_xlabel(\"Time Frame\")\n    plt.tight_layout(pad=0)\n\nplot_time_history(df=data_train, timeframes=['year',\"month\",\"day_of_week\",\"day_of_year\",\"hour\"], value = \"fare_amount\", color=\"blue\")\n    ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"21f44c3a16e5aafccc5b8c575411cdfd6e79b050"},"cell_type":"markdown","source":"**Baseline Model Development with Linear Regression**\n\n1. Separate features and target variables (Fare)\n2. Feature scaling : Scale the data of passenger, year, distance, hour, day\n3. Data will be split into two sets primarily training, validation and testing data.\n4. Baseling model with sklearn"},{"metadata":{"trusted":true,"_uuid":"842616e8661f6be569b923641bccb73008451ce8"},"cell_type":"code","source":"# Convert LAT/LONG to radians after plotting is done\ndata_train['pickup_latitude']=deg2rad(data_train['pickup_latitude'])\ndata_train['pickup_longitude']=deg2rad(data_train['pickup_longitude'])\ndata_train['dropoff_longitude']=deg2rad(data_train['dropoff_longitude'])\ndata_train['dropoff_latitude']=deg2rad(data_train['dropoff_latitude'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f499f2ea6612263130cd1288fa26f90cc1d69847"},"cell_type":"code","source":"#data_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5224fd896f3018bcd136ea271f22d5d2ff05459b"},"cell_type":"code","source":"data_train.drop('pickup_datetime',1, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d9d4e7b95b261d6a86d3ce03359b3943ef859b45"},"cell_type":"code","source":"import seaborn as sns\n\ncm = data_train.corr()\nplt.figure(figsize=(18, 18))\nsns.heatmap(cm, xticklabels=data_train.columns, yticklabels=data_train.columns, annot=True, cmap=\"YlGnBu\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"25879fb06cc54eb0b043ca629478a43a8cafee8e"},"cell_type":"code","source":"# Separate features from target variable\nfare = data_train['fare_amount']\nfeatures = data_train.drop('fare_amount', axis = 1)\n\n# Drop columns of LAT, LONG, Month, Day of the week, day of the year as they do not influence the fare but demand\n# Hour is dropped because it is transformed into peak hour or late night\n#features.drop('pickup_latitude',1, inplace=True)\n#features.drop('pickup_longitude',1, inplace=True)\n#features.drop('dropoff_latitude',1, inplace=True)\n#features.drop('dropoff_longitude',1, inplace=True)\n#features.drop('pickup_datetime',1, inplace=True)\nfeatures.drop('day_of_year',1, inplace=True)\nfeatures.drop('day_of_week',1, inplace=True)\nfeatures.drop('month',1, inplace=True)\nfeatures.drop('hour',1, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"74c268dee8885bfb910d634e75b9a69cfaf77e8a"},"cell_type":"code","source":"features.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d60dbf9e3e35c1f440f12240235476f9cd78763e"},"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\n\nscaler = MinMaxScaler()\nfeatures_scaled = scaler.fit_transform(features)\n\n#print(features_scaled[0])\n\nscaled_df = pd.DataFrame(features_scaled, columns=['pickup_longitude','pickup_latitude','dropoff_longitude','dropoff_latitude',\\\n                                             'passenger_count','distance_km','bearing','year','peak','night_hour',\n                                             'JFK_airport','NEW_airport'])\n#scaled_df.head()\nscaled_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f1cbd32d10c65f0dc6ed30eb1372c7836fea04be"},"cell_type":"code","source":"# Data splitting with sklearn train_test_split and 10% of the data is reserved for testing\nfrom sklearn.model_selection import train_test_split\n\nX_train1, X_hold, y_train1, y_hold = train_test_split(features_scaled, \n                                                    fare, \n                                                    test_size = 0.1, \n                                                    random_state = 42)\n\n# Show the results of the split\nprint(\"Training set has {} samples.\".format(X_train1.shape[0]))\nprint(\"Holdout set has {} samples.\".format(X_hold.shape[0]))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d70a584ac2c67df886b7ebf7956cee991f92cbee"},"cell_type":"code","source":"# Data splitting with sklearn train_test_split\nfrom sklearn.model_selection import train_test_split\n\nX_train, X_val, y_train, y_val = train_test_split(X_train1, \n                                                    y_train1, \n                                                    test_size = 0.2, \n                                                    random_state = 42)\n\n# Show the results of the split\nprint(\"Training set has {} samples.\".format(X_train.shape[0]))\nprint(\"Validation set has {} samples.\".format(X_val.shape[0]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f3abe5f5b46613119bc7acdbae05a1974febbd8f","scrolled":true},"cell_type":"code","source":"from sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.linear_model import LinearRegression\nfrom math import sqrt\n\nReg_model = LinearRegression()\nReg_model.fit(X_train, y_train)\ny_pred = Reg_model.predict(X_hold)\n\nerror = math.sqrt(mean_squared_error(y_pred,y_hold))\nprint (f'Root mean squared error is {error}')\n\nerror = mean_absolute_error(y_pred,y_hold)\nprint (f'Mean absolute error is {error}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3d9d1a45c55bda3f5869c2c85ee03c0969af71cb"},"cell_type":"code","source":"print ('Coefficients of the linear model are',Reg_model.coef_)\nprint('Intercept of the model is',Reg_model.intercept_)\nprint (f'R2 Score of the model is {Reg_model.score(X_hold, y_hold)}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c802c3a9b8487c076fe19dee43eda1bf74b5eb79"},"cell_type":"code","source":"#Reference https://github.com/jeffheaton/t81_558_deep_learning/blob/master/t81_558_class09_regularization.ipynb\n\n# Simple function to evaluate the coefficients of a regression\n%matplotlib inline    \nfrom IPython.display import display, HTML    \n\ndef report_coef(names,coef,intercept):\n    r = pd.DataFrame( { 'coef': coef, 'positive': coef>=0  }, index = names )\n    r = r.sort_values(by=['coef'])\n    display(r)\n    print(\"Intercept: {}\".format(intercept))\n    r['coef'].plot(kind='barh', color=r['positive'].map({True: 'b', False: 'r'}))\n    \nnames=['pickup_longitude','pickup_latitude','dropoff_longitude','dropoff_latitude',\\\n                                             'passenger_count','distance_km','bearing','year','peak','night_hour',\n                                             'JFK_airport','NEW_airport']   \nreport_coef(\n  names,\n  Reg_model.coef_,\n  Reg_model.intercept_)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8c68a16b8287e7532e54d7dae486cd64e5ff91d0"},"cell_type":"code","source":"plt.scatter(y_hold, y_pred)\nx = np.linspace(0, 100, 100)\ny = x\nplt.plot(x, y)\nplt.show()\n\n# reference:https://www.kaggle.com/andyxie/beginner-scikit-learn-linear-regression-tutorial","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4a931b82d3d8ef785fe880843d5509422eb4c5b9"},"cell_type":"markdown","source":"**Predicting Fares with LightGBM**\n\n"},{"metadata":{"trusted":true,"_uuid":"e5dfabca86bd0e86bf0a9d9d01409be73d3476f7"},"cell_type":"code","source":"params = {\n        'boosting_type':'gbdt',\n        'objective': 'regression',\n        'nthread': 4,\n        'num_leaves': 31,\n        'learning_rate': 0.05,\n        'max_depth': -1,\n        'subsample': 0.8,\n        'bagging_fraction' : 1,\n        'max_bin' : 5000 ,\n        'bagging_freq': 20,\n        'colsample_bytree': 0.6,\n        'metric': 'rmse',\n        'min_split_gain': 0.5,\n        'min_child_weight': 1,\n        'min_child_samples': 10,\n        'scale_pos_weight':1,\n        'zero_as_missing': True,\n        'seed':0,\n        'num_rounds':50000\n    }\n\ntrain_set = lgbm.Dataset(X_train, y_train, silent = False)\nvalid_set = lgbm.Dataset(X_val, y_val, silent = False)\nlgbm_model = lgbm.train(params=params, train_set=train_set, num_boost_round=10000, early_stopping_rounds=500, verbose_eval=500,valid_sets=valid_set)\n\nprediction = lgbm_model.predict(X_hold, num_iteration = lgbm_model.best_iteration)\nerror = math.sqrt(mean_squared_error(prediction,y_hold))\nprint (f'Root mean squared error is {error}')\n\nerror = mean_absolute_error(prediction,y_hold)\nprint (f'Mean absolute error is {error}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"68c94a1ee40692d004534fd88f307976cef645d7"},"cell_type":"code","source":"feature_imp = pd.DataFrame(sorted(zip(lgbm_model.feature_importance(),names)), columns=['Value','Feature'])\n\nplt.figure(figsize=(20, 10))\nsns.barplot(x=\"Value\", y=\"Feature\", data=feature_imp.sort_values(by=\"Value\", ascending=False))\nplt.title('LightGBM Features (avg over folds)')\nplt.tight_layout()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c7f4b67c9615cc6b680eb4f1bb2e63c3dbfa5815"},"cell_type":"markdown","source":"**Predicting Fares with Deep Learning Model with Tensorflow and Keras\n**\n\nFollowing will be parameters of the model used for tuning the model\n\n* Plot validation loss curve\n* Learning rate\n* Number of deep layers\n* Number of neurons in layers\n* Add dropout layers\n* Change initial values\n* Batch gradient descent for large data"},{"metadata":{"trusted":true,"_uuid":"0e604ff3fbdd238245f1b269d9a0f1140dc0c50e"},"cell_type":"code","source":"from keras import optimizers\nfrom keras.callbacks import ModelCheckpoint\nfrom keras.callbacks import ReduceLROnPlateau\nfrom keras.layers.normalization import BatchNormalization\nfrom keras.layers import Dropout\nfrom keras import regularizers\n\n\n# Build the neural network\nmodel = Sequential()\n\nmodel.add(Dense(256,input_dim=X_train.shape[1],activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.2))\n\nmodel.add(Dense(128,activation='relu',kernel_regularizer=regularizers.l2(0.01)))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.2))\n\nmodel.add(Dense(64,activation='relu',kernel_regularizer=regularizers.l2(0.01)))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.2))\n\nmodel.add(Dense(32,kernel_regularizer=regularizers.l2(0.01),\n                activity_regularizer=regularizers.l1(0.01),activation='relu'))\n\nmodel.add(Dense(1))\n\nsgd = optimizers.SGD(lr=0.001, decay=1e-4, momentum=0.9, nesterov=True)\nadam = optimizers.Adam(lr=0.001, beta_1=0.9, beta_2=0.999, epsilon=None, decay=0.0, amsgrad=False)\n\nmodel.compile(loss='mean_squared_error', optimizer=adam)\n\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3b5e5a13557ec63b9513ff360eef87ea5e5caba4","scrolled":true},"cell_type":"code","source":"from keras.callbacks import LearningRateScheduler\n\nmonitor = EarlyStopping(monitor='val_loss',min_delta=1e-3, patience=5, verbose=0, mode='auto')\n#reduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=10, min_lr=0.001,cooldown=5)\n\ndef step_decay(epoch):\n    initial_lrate = 0.001\n    drop = 0.5\n    epochs_drop = 5.0\n    lrate = initial_lrate * math.pow(drop, math.floor((1+epoch)/epochs_drop))\n    return lrate\n\nlrate = LearningRateScheduler(step_decay)\n\nhistory = model.fit(X_train,y_train,validation_data=(X_val,y_val), callbacks=[monitor], batch_size = 250, verbose=1, epochs=1000)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c709f935ab01ec222f93b78c5d6105fc6c1beb6e"},"cell_type":"code","source":"plt.plot(history.history['loss'], label='Training Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('History of Loss')\nplt.xlabel('Number of epochs')\nplt.ylabel('loss')\nplt.legend() ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"00040a6cae939c0307c60b68f9287b3897e0ba27","scrolled":true},"cell_type":"code","source":"# Making predictions\nprediction = model.predict(X_hold)\n\n# Check how good are predictions with RMS calculation\nRMS = np.sqrt(metrics.mean_squared_error(prediction,y_hold))\n\nprint(f'Model has RMS of : {RMS}')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bf89439c7b77e5ba817f281ee0f53214078996a6"},"cell_type":"markdown","source":"Optimzation of Hyperparameters\nIn this section, following hyperparameters are tuned to improve the accuracy of the model\n\n* Batch size: Ranges from 250-500\n* Dropout:  0-0.2"},{"metadata":{"trusted":true,"_uuid":"c1ef6d4a3f9274ff1831e315a6a1c5a3e4f69d5a"},"cell_type":"code","source":"def CreatDeepNetwork(dropout_rate=0.0):\n    # Build the neural network\n    Gmodel = Sequential()\n    Gmodel.add(Dense(256,input_dim=X_train.shape[1],activation='relu'))\n    Gmodel.add(BatchNormalization())\n    Gmodel.add(Dropout(dropout_rate))\n    \n    Gmodel.add(Dense(128,activation='relu',kernel_regularizer=regularizers.l2(0.01)))\n    Gmodel.add(BatchNormalization())\n    Gmodel.add(Dropout(dropout_rate))\n\n    Gmodel.add(Dense(64,activation='relu',kernel_regularizer=regularizers.l2(0.01)))\n    Gmodel.add(BatchNormalization())\n    Gmodel.add(Dropout(dropout_rate))\n    \n    Gmodel.add(Dense(32,kernel_regularizer=regularizers.l2(0.01),\n                activity_regularizer=regularizers.l1(0.01),activation='relu'))\n    Gmodel.add(Dense(1))\n    adam = optimizers.Adam(lr=0.001, beta_1=0.9, beta_2=0.999, epsilon=None, decay=0.0, amsgrad=False)\n\n    Gmodel.compile(loss='mean_squared_error', optimizer=adam)\n    return Gmodel\n#model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"676c7bdd34d0a03e092079745012d8cae7c720f7"},"cell_type":"code","source":"from keras.wrappers.scikit_learn import KerasRegressor\nfrom sklearn.metrics import make_scorer\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.model_selection import GridSearchCV\n\nneural_network = KerasRegressor(build_fn=CreatDeepNetwork, verbose=0)\nscoring_fnc = make_scorer(mean_squared_error,greater_is_better=False)\nbatch_size = [250, 500, 750]\ndropout_rate =[0.0, 0.1, 0.2]\n\n# Create hyperparameter options\nhyperparameters = dict(batch_size=batch_size,dropout_rate=dropout_rate)\n\ngrid = GridSearchCV(estimator=neural_network, param_grid=hyperparameters, cv=None,scoring= scoring_fnc)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"60e374bddbaf2a749d55f7f711e24712fc641b7b"},"cell_type":"code","source":"#Fit grid search\ngrid_result = grid.fit(X_train,y_train)\n\nprint(\"Best: %f using %s\" % (grid_result.best_score_, grid_result.best_params_))\nmeans = grid_result.cv_results_['mean_test_score']\nstds = grid_result.cv_results_['std_test_score']\nparams = grid_result.cv_results_['params']\nfor mean, stdev, param in zip(means, stds, params):\n    print(\"%f (%f) with: %r\" % (mean, stdev, param))","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}