{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\n\nimport xgboost as xgb\nfrom sklearn.metrics import mean_squared_error as mse\nfrom sklearn.model_selection import train_test_split\n\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import StandardScaler\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-10T19:30:41.559729Z","iopub.execute_input":"2022-08-10T19:30:41.560143Z","iopub.status.idle":"2022-08-10T19:30:42.211786Z","shell.execute_reply.started":"2022-08-10T19:30:41.560058Z","shell.execute_reply":"2022-08-10T19:30:42.210469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df =  pd.read_csv('../input/new-york-city-taxi-fare-prediction/train.csv', nrows = 10_000_000)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:30:42.213382Z","iopub.execute_input":"2022-08-10T19:30:42.213749Z","iopub.status.idle":"2022-08-10T19:31:09.069578Z","shell.execute_reply.started":"2022-08-10T19:30:42.213712Z","shell.execute_reply":"2022-08-10T19:31:09.067751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('../input/new-york-city-taxi-fare-prediction/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:31:09.075239Z","iopub.execute_input":"2022-08-10T19:31:09.076172Z","iopub.status.idle":"2022-08-10T19:31:09.105135Z","shell.execute_reply.started":"2022-08-10T19:31:09.076122Z","shell.execute_reply":"2022-08-10T19:31:09.103772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### We don't need key column for the training dataset","metadata":{}},{"cell_type":"code","source":"train_df.drop('key', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:31:09.109092Z","iopub.execute_input":"2022-08-10T19:31:09.109642Z","iopub.status.idle":"2022-08-10T19:31:09.742273Z","shell.execute_reply.started":"2022-08-10T19:31:09.109591Z","shell.execute_reply":"2022-08-10T19:31:09.740959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### View and explore the data, check if there are any null values or values that seem out of place","metadata":{}},{"cell_type":"code","source":"train_df.head()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.dtypes","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isna().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.float_format', lambda x: '%.5f' % x)\ntrain_df.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### We can see that the minimum fare amount is negative, which can't be correct.\n#### Also, the longitudes and latitudes have to be in between [-180, 180] and [-90, 90] respectively.\n#### Let's see how many of the values are incorrect.","metadata":{}},{"cell_type":"code","source":"(train_df['fare_amount'] < 0).sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"((train_df['pickup_longitude'] < -180) | (train_df['pickup_longitude'] > 180)).sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"((train_df['pickup_latitude'] < -90) | (train_df['pickup_latitude'] > 90)).sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### We also have to check that the passenger count is between 1 and 6","metadata":{}},{"cell_type":"code","source":"((train_df['passenger_count'] == 0) | (train_df['passenger_count'] > 6)).sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(train_df['passenger_count'] > 6).sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### There are quite a few entries with passenger count equal to 0, let's check if the test set contains any entries with 0 passenger count.","metadata":{}},{"cell_type":"code","source":"print(\"Train Passenger count equals 0: \", (train_df['passenger_count'] == 0).sum())\nprint(\"Test Passenger count equals 0: \", (test_df['passenger_count'] == 0).sum())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Seeing as there are no entries in the test set with passenger count equal to 0, I'll drop them from the train set in clean_df method below","metadata":{}},{"cell_type":"markdown","source":"### Let's explore some of the data visually","metadata":{}},{"cell_type":"markdown","source":"#### Check to see how the fare is distributed","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.histplot(train_df['fare_amount']);\nplt.title('Distribution of Fare Amount');","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### As the fare amounts over 200 are extremely rare, we can drop them as outliers","metadata":{}},{"cell_type":"markdown","source":"### We can see that only a small amount of values are labeled incorrectly, removing them from this huge dataset should have no impact.","metadata":{}},{"cell_type":"code","source":"# def clean_df(df):\n#     new_df = df[\n#         ((df.fare_amount > 0) & (df['fare_amount'] <= 200)) &\n#         ((df['pickup_longitude'] >= -180) & (df['pickup_longitude'] <= 180)) &  \n#         ((df['pickup_latitude'] >= -90) & (df['pickup_latitude'] <= 90)) &\n#         ((df['passenger_count'] > 0) & (df['passenger_count'] <= 6))\n#     ]\n    \n#     return new_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### I initally started with the dataframe cleaning method above, however adding in stricter restrictions to longitude and latitude had a significant improvement for the model\n### NYC longitude and latitude are (40.7, 70), so I limited the entries to be near the NYC.","metadata":{}},{"cell_type":"code","source":"def clean_df(df):\n    new_df = df[\n        ((df['fare_amount'] > 0) & (df['fare_amount'] <= 200)) &\n        ((df['pickup_longitude'] > -75) & (df['pickup_longitude'] < -73)) &  \n        ((df['pickup_latitude'] > 40) & (df['pickup_latitude'] < 42)) &\n        ((df['dropoff_longitude'] > -75) & (df['dropoff_longitude'] < -73)) &\n        ((df['dropoff_latitude'] > 40 & (df['dropoff_latitude'] < 42))) &\n        ((df['passenger_count'] > 0) & (df['passenger_count'] <= 6))\n    ]\n    \n    return new_df","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:31:09.744033Z","iopub.execute_input":"2022-08-10T19:31:09.744406Z","iopub.status.idle":"2022-08-10T19:31:09.751888Z","shell.execute_reply.started":"2022-08-10T19:31:09.744371Z","shell.execute_reply":"2022-08-10T19:31:09.750516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Here's the visualization of pickup coordinates before and after doing the data cleaning.\n### As is visible, before cleaning the data contains lots of absurd values.\n### Restricting the coordinates helps the model to learn","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12,8))\nsns.scatterplot(x=train_df['pickup_longitude'], y=train_df['pickup_latitude'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Before:\", len(train_df))\ntrain_df = clean_df(train_df)\nprint(\"After:\", len(train_df))\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:31:09.753692Z","iopub.execute_input":"2022-08-10T19:31:09.754091Z","iopub.status.idle":"2022-08-10T19:31:10.648338Z","shell.execute_reply.started":"2022-08-10T19:31:09.754056Z","shell.execute_reply":"2022-08-10T19:31:10.647112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,8))\nsns.scatterplot(x=train_df['pickup_longitude'], y=train_df['pickup_latitude'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Check to see if there's correlation between the number of passengers and the fare amount","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.scatterplot(x=train_df['passenger_count'], y=train_df['fare_amount'])\nplt.xlabel('Number of Passengers')\nplt.ylabel('Fare Amount')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### The fare amount seems to decrease with the increased number of passengers ","metadata":{}},{"cell_type":"code","source":"train_df.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineering\n#### An important metric for the fare amount is the distance travelled\n#### Since the earth is almost a sphere, we'll calculate the distances between two points using the haversine formula and add it to our data as a column","metadata":{}},{"cell_type":"code","source":"# def haversine_distance(lat_p, long_p, lat_d, long_d):\n#     R_earth = 6371  #radius of earth in kilometers\n    \n#     lat_p = np.radians(lat_p)\n#     long_p = np.radians(long_p)\n#     lat_d = np.radians(lat_d)\n#     long_d = np.radians(long_d)\n    \n#     long_diff = long_d - long_p\n#     lat_diff = lat_d - lat_p        \n    \n#     #a = sin²((φB - φA)/2) + cos φA . cos φB . sin²((λB - λA)/2)\n#     a = np.sin(lat_diff / 2.0) ** 2 + np.cos(lat_p) * np.cos(lat_d) * np.sin(long_diff / 2.0) ** 2\n    \n#     return 2 * R_earth * np.arcsin(np.sqrt(a))\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lat_p = train_df['pickup_latitude']\n# long_p = train_df['pickup_longitude']\n# lat_d = train_df['dropoff_latitude']\n# long_d = train_df['dropoff_longitude']\n\n# train_df['haversine_dist'] = haversine_distance(lat_p, long_p, lat_d, long_d)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### After experimenting for a while, I found that the manhattan distance formula performs better than the haversine, so I'll be using this for now","metadata":{}},{"cell_type":"code","source":"def manhattan_dist(lat_p, long_p, lat_d, long_d):  \n    distance = np.abs(lat_d - lat_p) + np.abs(long_d - long_p)\n    \n    return distance","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:31:10.649738Z","iopub.execute_input":"2022-08-10T19:31:10.650100Z","iopub.status.idle":"2022-08-10T19:31:10.655344Z","shell.execute_reply.started":"2022-08-10T19:31:10.650066Z","shell.execute_reply":"2022-08-10T19:31:10.654425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Let's extract the information from datetime object, as year, month, day and hour could all be important factors in predicting the fare amount","metadata":{}},{"cell_type":"code","source":"def add_datetime_info(df, transform_datetime=False):\n    if transform_datetime:\n        df['pickup_datetime'] = pd.to_datetime(df['pickup_datetime'], format=\"%Y-%m-%d %H:%M:%S UTC\")\n    \n    df['hour'] = df['pickup_datetime'].dt.hour\n    df['day'] = df['pickup_datetime'].dt.day\n    df['month'] = df['pickup_datetime'].dt.month\n    df['year'] = df['pickup_datetime'].dt.year\n#     df['weekday'] = df['pickup_datetime'].dt.weekday # removing this since it's the least important feature\n    df.drop('pickup_datetime', axis=1, inplace=True)\n        ","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:31:10.656564Z","iopub.execute_input":"2022-08-10T19:31:10.657219Z","iopub.status.idle":"2022-08-10T19:31:10.669006Z","shell.execute_reply.started":"2022-08-10T19:31:10.657152Z","shell.execute_reply":"2022-08-10T19:31:10.668041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### The intuition for adding the distances to the airports comes from this great [notebook](https://www.kaggle.com/code/breemen/nyc-taxi-fare-data-exploration)  ","metadata":{}},{"cell_type":"code","source":"def add_airport_info(df):\n#     nyc = (40.7141667, -74.0063889)\n#     jfk = (-73.7822222222, 40.6441666667)\n#     ewr = (-74.175, 40.69)\n#     lgr = (-73.87, 40.77)\n    \n    nyc = (-74.0063889, 40.7141667)\n    jfk = (40.6441666667, -73.7822222222)\n    ewr = (40.69, -74.175)\n    lgr = (40.77, -73.87)\n    \n    df['distance_to_center'] = manhattan_dist(nyc[0], nyc[1], df['pickup_latitude'], df['pickup_longitude'])\n    df['pickup_distance_to_jfk'] = manhattan_dist(jfk[0], jfk[1], df['pickup_latitude'], df['pickup_longitude'])\n    df['dropoff_distance_to_jfk'] = manhattan_dist(jfk[0], jfk[1], df['dropoff_latitude'], df['dropoff_longitude'])\n    df['pickup_distance_to_ewr'] = manhattan_dist(ewr[0], ewr[1], df['pickup_latitude'], df['pickup_longitude'])\n    df['dropoff_distance_to_ewr'] = manhattan_dist(ewr[0], ewr[1],df['dropoff_latitude'], df['dropoff_longitude'])\n    df['pickup_distance_to_lgr'] = manhattan_dist(lgr[0], lgr[1], df['pickup_latitude'], df['pickup_longitude'])\n    df['dropoff_distance_to_lgr'] = manhattan_dist(lgr[0], lgr[1], df['dropoff_latitude'], df['dropoff_longitude'])\n\n    df['long_diff'] = df.dropoff_longitude - df.pickup_longitude\n    df['lat_diff'] = df.dropoff_latitude - df.pickup_latitude\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:31:54.476235Z","iopub.execute_input":"2022-08-10T19:31:54.477195Z","iopub.status.idle":"2022-08-10T19:31:54.486856Z","shell.execute_reply.started":"2022-08-10T19:31:54.477132Z","shell.execute_reply":"2022-08-10T19:31:54.485971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def transform(df, transform_datetime):\n    add_datetime_info(df, transform_datetime)\n    add_airport_info(df)\n    df['manhattan_dist'] = manhattan_dist(df['pickup_latitude'], df['pickup_longitude'], df['dropoff_latitude'], df['dropoff_longitude'])\n    return df\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:31:55.520137Z","iopub.execute_input":"2022-08-10T19:31:55.521271Z","iopub.status.idle":"2022-08-10T19:31:55.526749Z","shell.execute_reply.started":"2022-08-10T19:31:55.521227Z","shell.execute_reply":"2022-08-10T19:31:55.525787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We also need to change the pickup_datetime column from being an object to a datetime object.","metadata":{}},{"cell_type":"code","source":"# train_df['pickup_datetime'] =  pd.to_datetime(train_df['pickup_datetime'], format=\"%Y-%m-%d %H:%M:%S UTC\")\ntrain_df = transform(train_df, transform_datetime=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:31:59.561996Z","iopub.execute_input":"2022-08-10T19:31:59.562710Z","iopub.status.idle":"2022-08-10T19:32:38.511004Z","shell.execute_reply.started":"2022-08-10T19:31:59.562655Z","shell.execute_reply":"2022-08-10T19:32:38.509671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:38.512881Z","iopub.execute_input":"2022-08-10T19:32:38.513273Z","iopub.status.idle":"2022-08-10T19:32:38.544505Z","shell.execute_reply.started":"2022-08-10T19:32:38.513238Z","shell.execute_reply":"2022-08-10T19:32:38.543545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## I want to add some more Visualizations to see how the fare amount is changing with regards to different time periods.","metadata":{}},{"cell_type":"code","source":"def visualize_date_fare(df):\n#     date_objects = ['hour', 'day', 'weekday', 'month', 'year']\n    date_objects = ['hour', 'day', 'month', 'year']\n\n    for idx, obj in enumerate(date_objects):\n    #     print(\"IDX\", idx)\n    #     print(\"OBJ\", obj)\n        plt.figure(figsize=(10,6))\n        sns.barplot(x=df[obj], y=df['fare_amount'], ci=None) # setting ci=None dramatically decreases the time for plotting\n        print(plt.get_cmap())\n        plt.title('Average Fare Amount by ' + obj)\n        plt.ylabel('Fare Amount')\n        \n        plt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_date_fare(train_df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### There is a steady trend of increasing fare amounts by the year. We can see the fare amount also changes noticeably by the hour.\n### However, the fare amount doesn't seem to change that much from day to day.","metadata":{}},{"cell_type":"code","source":"def visualize_date_counts(df):\n    #     date_objects = ['hour', 'day', 'weekday', 'month', 'year']\n    date_objects = ['hour', 'day', 'month', 'year']\n    # fig, axes = plt.subplots(1, 5)\n\n    for obj in date_objects:\n        plt.figure(figsize=(10,6))\n        sns.countplot(x=df[obj])\n        plt.ylabel('Count')\n        plt.title('Taxi Rides Count by ' + obj)\n        plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Graphs of the amount of taxi rides by time periods","metadata":{}},{"cell_type":"code","source":"visualize_date_counts(train_df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training","metadata":{}},{"cell_type":"markdown","source":"### XGBoost's feature importance showed that the passenger count was the lowest feature in terms of importance.\n### This aligns with the NYC government's official website, which states that [\"There is no extra charge for extra passengers\"](http://www1.nyc.gov/site/tlc/passengers/taxi-fare.page)\n### Bearing that in mind, I dropped the passenger count feature from the dataset.","metadata":{}},{"cell_type":"code","source":"# X = train_df.drop(['fare_amount', 'passenger_count'], axis=1)\n# y = train_df['fare_amount']\n\nX = train_df.drop(['fare_amount'], axis=1)\ny = train_df['fare_amount']","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:49.261914Z","iopub.execute_input":"2022-08-10T19:32:49.262904Z","iopub.status.idle":"2022-08-10T19:32:50.981008Z","shell.execute_reply.started":"2022-08-10T19:32:49.262839Z","shell.execute_reply":"2022-08-10T19:32:50.979755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X, y, random_state=42, test_size=0.05)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:51.264304Z","iopub.execute_input":"2022-08-10T19:32:51.265646Z","iopub.status.idle":"2022-08-10T19:32:55.815579Z","shell.execute_reply.started":"2022-08-10T19:32:51.265593Z","shell.execute_reply":"2022-08-10T19:32:55.813848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:55.819220Z","iopub.execute_input":"2022-08-10T19:32:55.820111Z","iopub.status.idle":"2022-08-10T19:32:55.842693Z","shell.execute_reply.started":"2022-08-10T19:32:55.820049Z","shell.execute_reply":"2022-08-10T19:32:55.841829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del(X)\ndel(y)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:55.843796Z","iopub.execute_input":"2022-08-10T19:32:55.844951Z","iopub.status.idle":"2022-08-10T19:32:55.858380Z","shell.execute_reply.started":"2022-08-10T19:32:55.844912Z","shell.execute_reply":"2022-08-10T19:32:55.857171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Commented because training takes too long during submit, commented on the results below","metadata":{}},{"cell_type":"markdown","source":"## Linear Regression","metadata":{}},{"cell_type":"code","source":"# from sklearn.linear_model import LinearRegression\n\n# reg_model = Pipeline((\n#         (\"standard_scaler\", StandardScaler()),\n#         (\"lin_reg\", LinearRegression()),\n#     ))\n# reg_model.fit(X_train, y_train)\n\n# y_train_pred = reg_model.predict(X_train)\n# y_val_pred = reg_model.predict(X_val)\n\n# print(\"Linear Regression Train RMSE: \", np.sqrt(mse(y_train, y_train_pred)))\n# print(\"Linear Regression Validation RMSE: \", np.sqrt(mse(y_val, y_val_pred)))\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Random Forest","metadata":{}},{"cell_type":"code","source":"# from sklearn.ensemble import RandomForestRegressor\n\n# reg_model = Pipeline((\n#         (\"standard_scaler\", StandardScaler()),\n#         (\"lin_reg\", RandomForestRegressor(max_depth=2, random_state=42, n_jobs=-1, verbose=True, n_estimators=100)),\n#     ))\n# reg_model.fit(X_train, y_train)\n\n# y_train_pred = reg_model.predict(X_train)\n# y_val_pred = reg_model.predict(X_val)\n\n# print(\"Random Forest Train RMSE: \", np.sqrt(mse(y_train, y_train_pred)))\n# print(\"Random Forest Validation RMSE: \", np.sqrt(mse(y_val, y_val_pred)))\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Extreme Gradient Boosting (XGBoost)","metadata":{}},{"cell_type":"markdown","source":"### Parameter tuning was taking too long, so I used parameters from [this](https://www.kaggle.com/code/btyuhas/bayesian-optimization-with-xgboost) notebook, played around for a while and these turned out to work pretty well.","metadata":{}},{"cell_type":"code","source":"def XGBoost(X_train,X_test,y_train,y_test):\n    dtrain = xgb.DMatrix(X_train,label=y_train)\n    dtest = xgb.DMatrix(X_test,label=y_test)\n\n    return xgb.train(params={'objective':'reg:linear','eval_metric':'rmse', 'max_depth':7, 'colsample_bytree':0.9, 'gamma':1}\n                    ,dtrain=dtrain,num_boost_round=400, \n                    early_stopping_rounds=30,evals=[(dtest,'test')])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:55.861776Z","iopub.execute_input":"2022-08-10T19:32:55.862533Z","iopub.status.idle":"2022-08-10T19:32:55.869666Z","shell.execute_reply.started":"2022-08-10T19:32:55.862486Z","shell.execute_reply":"2022-08-10T19:32:55.868837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_model = XGBoost(X_train, X_val, y_train, y_val)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:57.199540Z","iopub.execute_input":"2022-08-10T19:32:57.200429Z","iopub.status.idle":"2022-08-10T19:34:01.190433Z","shell.execute_reply.started":"2022-08-10T19:32:57.200387Z","shell.execute_reply":"2022-08-10T19:34:01.188973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train_pred = xgb_model.predict(xgb.DMatrix(X_train), ntree_limit = xgb_model.best_iteration)\ny_val_pred = xgb_model.predict(xgb.DMatrix(X_val), ntree_limit = xgb_model.best_iteration)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Train set error: \", np.sqrt(mse(y_train, y_train_pred)))\nprint(\"Validation set error: \", np.sqrt(mse(y_val, y_val_pred)))\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## As expected, the linear model performs the worst on both train and validation sets. The XGBoost outperforms the Random Forest model, however I haven't done any parameter tuning on it so the actual results might be closer.","metadata":{}},{"cell_type":"code","source":"#Read and preprocess test set\ntest_df =  pd.read_csv('../input/new-york-city-taxi-fare-prediction/test.csv')\n# test_df['pickup_datetime'] = pd.to_datetime(test_df['pickup_datetime'], format=\"%Y-%m-%d %H:%M:%S UTC\")\ntest_df = transform(test_df, transform_datetime=True)\n# test_df['manhattan_dist'] = manhattan_dist(test_df['pickup_latitude'], test_df['pickup_longitude'], \n#                                    test_df['dropoff_latitude'] , test_df['dropoff_longitude'])\n\n\n\ntest_key = test_df['key']\n# x_pred = test_df.drop(columns=['key', 'passenger_count'])\nx_pred = test_df.drop(columns=['key'])\n\n# Predict from test set\nprediction = xgb_model.predict(xgb.DMatrix(x_pred))\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(x_pred)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction = prediction.round(2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\n        \"key\": test_key,\n        \"fare_amount\": prediction\n})\n\n# submission.to_csv('/kaggle/working/XGBSubmission.csv',index=False)\nsubmission.to_csv('submission.csv',index=False)\n        ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('submission.csv').head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Plotting Feature Importance')\nfig, ax = plt.subplots(figsize=(12,16))\nxgb.plot_importance(xgb_model, height=0.7, ax=ax)\nax.grid(False)\nplt.title(\"XGBoost - Feature Importance\", fontsize=14)\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Possible Improvements:\n* ### Train with more data, currently I use 10 million examples\n* ### Tune hyperparameters using Cross Validation.\n* ### Could take real life New York taxi fares in account, for example currently initial charge for taxis in NYC is 2.5$.\n* ### Could take holidays and special occations in account, since the tax fares are higher on those days","metadata":{}}]}