{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-10T19:32:00.265258Z","iopub.execute_input":"2022-08-10T19:32:00.265699Z","iopub.status.idle":"2022-08-10T19:32:00.286465Z","shell.execute_reply.started":"2022-08-10T19:32:00.265665Z","shell.execute_reply":"2022-08-10T19:32:00.284996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nimport plotly.express as px\nimport plotly.graph_objects as go","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:00.432843Z","iopub.execute_input":"2022-08-10T19:32:00.433297Z","iopub.status.idle":"2022-08-10T19:32:00.439150Z","shell.execute_reply.started":"2022-08-10T19:32:00.433254Z","shell.execute_reply":"2022-08-10T19:32:00.437904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df =  pd.read_csv('../input/new-york-city-taxi-fare-prediction/train.csv', nrows = 1_000_000)\ntest_df = pd.read_csv('../input/new-york-city-taxi-fare-prediction/test.csv')\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:00.736517Z","iopub.execute_input":"2022-08-10T19:32:00.737812Z","iopub.status.idle":"2022-08-10T19:32:03.358506Z","shell.execute_reply.started":"2022-08-10T19:32:00.737746Z","shell.execute_reply":"2022-08-10T19:32:03.357125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:03.360789Z","iopub.execute_input":"2022-08-10T19:32:03.361497Z","iopub.status.idle":"2022-08-10T19:32:03.484385Z","shell.execute_reply.started":"2022-08-10T19:32:03.361458Z","shell.execute_reply":"2022-08-10T19:32:03.482776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['pickup_datetime'] =  pd.to_datetime(train_df['pickup_datetime'], format='%Y-%m-%d %H:%M:%S %Z')\ntest_df['pickup_datetime'] =  pd.to_datetime(test_df['pickup_datetime'], format='%Y-%m-%d %H:%M:%S %Z')","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:03.486456Z","iopub.execute_input":"2022-08-10T19:32:03.486876Z","iopub.status.idle":"2022-08-10T19:32:10.248353Z","shell.execute_reply.started":"2022-08-10T19:32:03.486839Z","shell.execute_reply":"2022-08-10T19:32:10.241801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe().applymap('{:,.2f}'.format)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.249628Z","iopub.status.idle":"2022-08-10T19:32:10.250120Z","shell.execute_reply.started":"2022-08-10T19:32:10.249873Z","shell.execute_reply":"2022-08-10T19:32:10.249923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.251290Z","iopub.status.idle":"2022-08-10T19:32:10.251695Z","shell.execute_reply.started":"2022-08-10T19:32:10.251504Z","shell.execute_reply":"2022-08-10T19:32:10.251524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There is a difference in value ranges of variables between test data and train data. Minimum and maximum values of some variables in the train data points to the existence of outliers. For more accurate prediction, we can drop the values outside the test data's domain.","metadata":{}},{"cell_type":"code","source":"train_df.pickup_datetime.min(), train_df.pickup_datetime.max()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.253255Z","iopub.status.idle":"2022-08-10T19:32:10.253662Z","shell.execute_reply.started":"2022-08-10T19:32:10.253464Z","shell.execute_reply":"2022-08-10T19:32:10.253484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.pickup_datetime.min(), test_df.pickup_datetime.max()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.254856Z","iopub.status.idle":"2022-08-10T19:32:10.255285Z","shell.execute_reply.started":"2022-08-10T19:32:10.255080Z","shell.execute_reply":"2022-08-10T19:32:10.255099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can see, value ranges for datetime are the same for train and test dataset.","metadata":{}},{"cell_type":"code","source":"train_df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.256595Z","iopub.status.idle":"2022-08-10T19:32:10.257060Z","shell.execute_reply.started":"2022-08-10T19:32:10.256793Z","shell.execute_reply":"2022-08-10T19:32:10.256812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.258027Z","iopub.status.idle":"2022-08-10T19:32:10.258471Z","shell.execute_reply.started":"2022-08-10T19:32:10.258283Z","shell.execute_reply":"2022-08-10T19:32:10.258302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Test data does not contain null values. The number of rows with missing values for the training data is not much of significance, so we can drop these rows.","metadata":{}},{"cell_type":"code","source":"train_df = train_df.dropna()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.259628Z","iopub.status.idle":"2022-08-10T19:32:10.260008Z","shell.execute_reply.started":"2022-08-10T19:32:10.259810Z","shell.execute_reply":"2022-08-10T19:32:10.259828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['passenger_count'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.261097Z","iopub.status.idle":"2022-08-10T19:32:10.261478Z","shell.execute_reply.started":"2022-08-10T19:32:10.261291Z","shell.execute_reply":"2022-08-10T19:32:10.261309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def outliers(df,col,x,y):\n    a = df[~((df[col] > x) & (df[col] <= y))][col].value_counts()\n    df_value_counts = pd.DataFrame(a)\n    df_value_counts = df_value_counts.reset_index()\n    df_value_counts.columns = ['unique_values', 'counts']\n    return df_value_counts","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.262475Z","iopub.status.idle":"2022-08-10T19:32:10.262830Z","shell.execute_reply.started":"2022-08-10T19:32:10.262655Z","shell.execute_reply":"2022-08-10T19:32:10.262672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pass_outliers = outliers(train_df,'passenger_count',0,6)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.264086Z","iopub.status.idle":"2022-08-10T19:32:10.264458Z","shell.execute_reply.started":"2022-08-10T19:32:10.264282Z","shell.execute_reply":"2022-08-10T19:32:10.264300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#fig = px.histogram(train_df, x='passenger_count')\n#fig.add_trace(go.Scatter(x = pass_outliers['unique_values'],y = pass_outliers['counts'],mode=\"markers\"))\nsns.histplot(data=train_df, x='passenger_count',bins=200)\nsns.scatterplot(x=pass_outliers['unique_values'],y = pass_outliers['counts'])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.265476Z","iopub.status.idle":"2022-08-10T19:32:10.265856Z","shell.execute_reply.started":"2022-08-10T19:32:10.265671Z","shell.execute_reply":"2022-08-10T19:32:10.265691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#fig = px.scatter(x = train_df['passenger_count'], y = train_df['fare_amount'])\n#fig.show()\nsns.scatterplot(x = train_df['passenger_count'],y = train_df['fare_amount'])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.267033Z","iopub.status.idle":"2022-08-10T19:32:10.267400Z","shell.execute_reply.started":"2022-08-10T19:32:10.267224Z","shell.execute_reply":"2022-08-10T19:32:10.267242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a = train_df['fare_amount'].value_counts()\ndf_value_counts = pd.DataFrame(a)\ndf_value_counts = df_value_counts.reset_index()\ndf_value_counts.columns = ['unique_values', 'counts']","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.268423Z","iopub.status.idle":"2022-08-10T19:32:10.268784Z","shell.execute_reply.started":"2022-08-10T19:32:10.268605Z","shell.execute_reply":"2022-08-10T19:32:10.268623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fare_outliers = outliers(train_df,'fare_amount',0,300)\nfare_outliers","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.269975Z","iopub.status.idle":"2022-08-10T19:32:10.270382Z","shell.execute_reply.started":"2022-08-10T19:32:10.270190Z","shell.execute_reply":"2022-08-10T19:32:10.270210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.scatterplot(x=df_value_counts['unique_values'],y = df_value_counts['counts'])\nsns.scatterplot(x=fare_outliers['unique_values'],y = fare_outliers['counts'])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.271842Z","iopub.status.idle":"2022-08-10T19:32:10.272275Z","shell.execute_reply.started":"2022-08-10T19:32:10.272079Z","shell.execute_reply":"2022-08-10T19:32:10.272100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in ['pickup_longitude', 'pickup_latitude', 'dropoff_longitude', 'dropoff_latitude']:\n    f, axes = plt.subplots(1, 2)\n    sns.kdeplot(train_df[i],ax=axes[0])\n    sns.kdeplot(test_df[i],ax=axes[1])\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.274775Z","iopub.status.idle":"2022-08-10T19:32:10.275251Z","shell.execute_reply.started":"2022-08-10T19:32:10.275037Z","shell.execute_reply":"2022-08-10T19:32:10.275059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def time_features(df):    \n    df['year'] = df['pickup_datetime'].apply(lambda x: x.year)\n    df['month'] = df['pickup_datetime'].apply(lambda x: x.month)\n    df['day'] = df['pickup_datetime'].apply(lambda x: x.day)\n    df['hour'] = df['pickup_datetime'].apply(lambda x: x.hour)\n    df['weekday'] = df['pickup_datetime'].apply(lambda x: x.weekday())\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.277058Z","iopub.status.idle":"2022-08-10T19:32:10.277573Z","shell.execute_reply.started":"2022-08-10T19:32:10.277362Z","shell.execute_reply":"2022-08-10T19:32:10.277384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean(df):\n    df = df[(df['pickup_longitude'] >= -75) & (df['pickup_longitude'] <= -72)]\n    df = df[(df['dropoff_longitude'] >= -75) & (df['dropoff_longitude'] <= -72)]\n    df = df[(df['pickup_latitude'] >= 40) & (df['pickup_latitude'] <= 42)]\n    df = df[(df['dropoff_latitude'] >= 40) & (df['dropoff_latitude'] <= 42)]\n    \n    df = df[(df['passenger_count'] > 0) & (df['passenger_count'] <= 6)]\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.278632Z","iopub.status.idle":"2022-08-10T19:32:10.279074Z","shell.execute_reply.started":"2022-08-10T19:32:10.278831Z","shell.execute_reply":"2022-08-10T19:32:10.278850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can clean the train dataset by removing the outliers using the ranges for test data. For coordinates, we are using data\nthat falls within the range of the coordinates for NYC. This will also help remove abnormally large and small values that were put\nby mistake. The outliers for passenger count will also be removed.","metadata":{"execution":{"iopub.status.busy":"2022-08-10T02:17:40.222186Z","iopub.execute_input":"2022-08-10T02:17:40.222672Z","iopub.status.idle":"2022-08-10T02:17:40.239315Z","shell.execute_reply.started":"2022-08-10T02:17:40.222631Z","shell.execute_reply":"2022-08-10T02:17:40.234610Z"}}},{"cell_type":"code","source":"train_df = time_features(train_df)\ntest_df = time_features(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.280525Z","iopub.status.idle":"2022-08-10T19:32:10.280941Z","shell.execute_reply.started":"2022-08-10T19:32:10.280723Z","shell.execute_reply":"2022-08-10T19:32:10.280742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = clean(train_df)\ntest_df = clean(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.282635Z","iopub.status.idle":"2022-08-10T19:32:10.283059Z","shell.execute_reply.started":"2022-08-10T19:32:10.282835Z","shell.execute_reply":"2022-08-10T19:32:10.282852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Non-positive values for taxi fare will be removed, as well as values that seems to be too high.","metadata":{"execution":{"iopub.status.busy":"2022-08-10T02:17:40.249707Z","iopub.status.idle":"2022-08-10T02:17:40.251306Z","shell.execute_reply.started":"2022-08-10T02:17:40.250834Z","shell.execute_reply":"2022-08-10T02:17:40.250882Z"}}},{"cell_type":"code","source":"train_df = train_df[(train_df['fare_amount'] > 0) & (train_df['fare_amount'] <= 300)]","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.284129Z","iopub.status.idle":"2022-08-10T19:32:10.284493Z","shell.execute_reply.started":"2022-08-10T19:32:10.284315Z","shell.execute_reply":"2022-08-10T19:32:10.284333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe().applymap('{:,.2f}'.format)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.286105Z","iopub.status.idle":"2022-08-10T19:32:10.286476Z","shell.execute_reply.started":"2022-08-10T19:32:10.286294Z","shell.execute_reply":"2022-08-10T19:32:10.286312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Useful variable for prediction will be the distance driven by taxi, and popular locations passenger might go to. JFK airport, the LaGuardia airport, the Newark airport, Times Square, Central Park, the Statue of Liberty, Grand Central, the MET museum, and the World Trade Center will be used. As for driven distance, great circle and geodesic distances will be added.","metadata":{"execution":{"iopub.status.busy":"2022-08-10T02:17:40.260043Z","iopub.status.idle":"2022-08-10T02:17:40.260558Z","shell.execute_reply.started":"2022-08-10T02:17:40.260324Z","shell.execute_reply":"2022-08-10T02:17:40.260352Z"}}},{"cell_type":"code","source":"import geopy.distance","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.290719Z","iopub.status.idle":"2022-08-10T19:32:10.291166Z","shell.execute_reply.started":"2022-08-10T19:32:10.290960Z","shell.execute_reply":"2022-08-10T19:32:10.290981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def geodesic_dist(df):\n    pickup_lat = df['pickup_latitude']\n    pickup_long = df['pickup_longitude']\n    dropoff_lat = df['dropoff_latitude']\n    dropoff_long = df['dropoff_longitude']\n    distance = geopy.distance.geodesic((pickup_lat, pickup_long), \n                                       (dropoff_lat, dropoff_long)).miles\n    try:\n        return distance\n    except ValueError:\n        return np.nan","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.292532Z","iopub.status.idle":"2022-08-10T19:32:10.293246Z","shell.execute_reply.started":"2022-08-10T19:32:10.293025Z","shell.execute_reply":"2022-08-10T19:32:10.293047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def circle_dist(df):\n    pickup_lat = df['pickup_latitude']\n    pickup_long = df['pickup_longitude']\n    dropoff_lat = df['dropoff_latitude']\n    dropoff_long = df['dropoff_longitude']\n    distance = geopy.distance.great_circle((pickup_lat, pickup_long), \n                                       (dropoff_lat, dropoff_long)).miles\n    #try:\n    return distance\n    #except ValueError:\n        #return np.nan","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.295139Z","iopub.status.idle":"2022-08-10T19:32:10.295551Z","shell.execute_reply.started":"2022-08-10T19:32:10.295356Z","shell.execute_reply":"2022-08-10T19:32:10.295376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['geodesic_dist'] = train_df.apply(lambda x: geodesic_dist(x), axis = 1 )","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.297065Z","iopub.status.idle":"2022-08-10T19:32:10.297824Z","shell.execute_reply.started":"2022-08-10T19:32:10.297603Z","shell.execute_reply":"2022-08-10T19:32:10.297625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['circle_dist'] = train_df.apply(lambda x: circle_dist(x), axis = 1 )","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.299138Z","iopub.status.idle":"2022-08-10T19:32:10.299536Z","shell.execute_reply.started":"2022-08-10T19:32:10.299345Z","shell.execute_reply":"2022-08-10T19:32:10.299364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def jfk_dist(df):\n    jfk_lat = 40.6413\n    jfk_long = -73.7781\n    dropoff_lat = df['dropoff_latitude']\n    dropoff_long = df['dropoff_longitude']\n    jfk_distance = geopy.distance.geodesic((dropoff_lat, dropoff_long), (jfk_lat, jfk_long)).miles\n    return jfk_distance\n\ndef lga_dist(df):\n    lga_lat = 40.7769\n    lga_long = -73.8740\n    dropoff_lat = df['dropoff_latitude']\n    dropoff_long = df['dropoff_longitude']\n    lga_distance = geopy.distance.geodesic((dropoff_lat, dropoff_long), (lga_lat, lga_long)).miles\n    return lga_distance\n\ndef ewr_dist(df):\n    ewr_lat = 40.6895\n    ewr_long = -74.1745\n    dropoff_lat = df['dropoff_latitude']\n    dropoff_long = df['dropoff_longitude']\n    ewr_distance = geopy.distance.geodesic((dropoff_lat, dropoff_long), (ewr_lat, ewr_long)).miles\n    return ewr_distance\n\ndef tsq_dist(df):\n    tsq_lat = 40.7580\n    tsq_long = -73.9855\n    dropoff_lat = df['dropoff_latitude']\n    dropoff_long = df['dropoff_longitude']\n    tsq_distance = geopy.distance.geodesic((dropoff_lat, dropoff_long), (tsq_lat, tsq_long)).miles\n    return tsq_distance\n\ndef cpk_dist(df):\n    cpk_lat = 40.7812\n    cpk_long = -73.9665\n    dropoff_lat = df['dropoff_latitude']\n    dropoff_long = df['dropoff_longitude']\n    cpk_distance = geopy.distance.geodesic((dropoff_lat, dropoff_long), (cpk_lat, cpk_long)).miles\n    return cpk_distance\n\ndef lib_dist(df):\n    lib_lat = 40.6892\n    lib_long = -74.0445\n    dropoff_lat = df['dropoff_latitude']\n    dropoff_long = df['dropoff_longitude']\n    lib_distance = geopy.distance.geodesic((dropoff_lat, dropoff_long), (lib_lat, lib_long)).miles\n    return lib_distance\n\ndef gct_dist(df):\n    gct_lat = 40.7527\n    gct_long = -73.9772\n    dropoff_lat = df['dropoff_latitude']\n    dropoff_long = df['dropoff_longitude']\n    gct_distance = geopy.distance.geodesic((dropoff_lat, dropoff_long), (gct_lat, gct_long)).miles\n    return gct_distance\n\ndef met_dist(df):\n    met_lat = 40.7794\n    met_long = -73.9632\n    dropoff_lat = df['dropoff_latitude']\n    dropoff_long = df['dropoff_longitude']\n    met_distance = geopy.distance.geodesic((dropoff_lat, dropoff_long), (met_lat, met_long)).miles\n    return met_distance\n\ndef wtc_dist(df):\n    wtc_lat = 40.7126\n    wtc_long = -74.0099\n    dropoff_lat = df['dropoff_latitude']\n    dropoff_long = df['dropoff_longitude']\n    wtc_distance = geopy.distance.geodesic((dropoff_lat, dropoff_long), (wtc_lat, wtc_long)).miles\n    return wtc_distance","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.305770Z","iopub.status.idle":"2022-08-10T19:32:10.306291Z","shell.execute_reply.started":"2022-08-10T19:32:10.306057Z","shell.execute_reply":"2022-08-10T19:32:10.306082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calc_dists(df): \n    df['jfk'] = df.apply(lambda x: jfk_dist(x), axis = 1 )\n    df['lga'] = df.apply(lambda x: lga_dist(x), axis = 1 )\n    df['ewr'] = df.apply(lambda x: ewr_dist(x), axis = 1 )\n    df['tsq'] = df.apply(lambda x: tsq_dist(x), axis = 1 )\n    df['cpk'] = df.apply(lambda x: cpk_dist(x), axis = 1 )\n    df['lib'] = df.apply(lambda x: lib_dist(x), axis = 1 )\n    df['gct'] = df.apply(lambda x: gct_dist(x), axis = 1 )\n    df['met'] = df.apply(lambda x: met_dist(x), axis = 1 )\n    df['wtc'] = df.apply(lambda x: wtc_dist(x), axis = 1 )\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.307939Z","iopub.status.idle":"2022-08-10T19:32:10.308776Z","shell.execute_reply.started":"2022-08-10T19:32:10.308250Z","shell.execute_reply":"2022-08-10T19:32:10.308279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"locs = ['jfk', 'lga', 'ewr', 'tsq', 'cpk','lib', 'gct', 'met', 'wtc']","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.310414Z","iopub.status.idle":"2022-08-10T19:32:10.311022Z","shell.execute_reply.started":"2022-08-10T19:32:10.310700Z","shell.execute_reply":"2022-08-10T19:32:10.310728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"distances = ['geodesic_dist','circle_dist']","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.312466Z","iopub.status.idle":"2022-08-10T19:32:10.313072Z","shell.execute_reply.started":"2022-08-10T19:32:10.312758Z","shell.execute_reply":"2022-08-10T19:32:10.312784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = calc_dists(train_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.314641Z","iopub.status.idle":"2022-08-10T19:32:10.315269Z","shell.execute_reply.started":"2022-08-10T19:32:10.314951Z","shell.execute_reply":"2022-08-10T19:32:10.314978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['geodesic_dist'] = test_df.apply(lambda x: geodesic_dist(x), axis = 1 )\ntest_df['circle_dist'] = test_df.apply(lambda x: circle_dist(x), axis = 1 )\ntest_df = calc_dists(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.316841Z","iopub.status.idle":"2022-08-10T19:32:10.317413Z","shell.execute_reply.started":"2022-08-10T19:32:10.317131Z","shell.execute_reply":"2022-08-10T19:32:10.317158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe().applymap('{:,.2f}'.format)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.318833Z","iopub.status.idle":"2022-08-10T19:32:10.319391Z","shell.execute_reply.started":"2022-08-10T19:32:10.319117Z","shell.execute_reply":"2022-08-10T19:32:10.319145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will be dropping columns we do not need.","metadata":{"execution":{"iopub.status.busy":"2022-08-10T02:23:55.367454Z","iopub.execute_input":"2022-08-10T02:23:55.368087Z","iopub.status.idle":"2022-08-10T02:23:55.375927Z","shell.execute_reply.started":"2022-08-10T02:23:55.368045Z","shell.execute_reply":"2022-08-10T02:23:55.374327Z"}}},{"cell_type":"code","source":"dropped_columns = ['pickup_longitude', 'pickup_latitude', \n                   'dropoff_longitude', 'dropoff_latitude', 'pickup_datetime']\ntrain = train_df.drop(dropped_columns, axis=1)\ntest = test_df.drop(dropped_columns, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.321337Z","iopub.status.idle":"2022-08-10T19:32:10.321912Z","shell.execute_reply.started":"2022-08-10T19:32:10.321599Z","shell.execute_reply":"2022-08-10T19:32:10.321626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.describe().applymap('{:,.2f}'.format)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.323218Z","iopub.status.idle":"2022-08-10T19:32:10.323785Z","shell.execute_reply.started":"2022-08-10T19:32:10.323477Z","shell.execute_reply":"2022-08-10T19:32:10.323506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe().applymap('{:,.2f}'.format)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.325313Z","iopub.status.idle":"2022-08-10T19:32:10.325680Z","shell.execute_reply.started":"2022-08-10T19:32:10.325500Z","shell.execute_reply":"2022-08-10T19:32:10.325517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictors = ['passenger_count', 'year', 'weekday','month','hour']","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.326774Z","iopub.status.idle":"2022-08-10T19:32:10.327160Z","shell.execute_reply.started":"2022-08-10T19:32:10.326974Z","shell.execute_reply":"2022-08-10T19:32:10.326992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in predictors:\n    sns.scatterplot(data = train, x = i, y = 'fare_amount')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.332300Z","iopub.status.idle":"2022-08-10T19:32:10.333199Z","shell.execute_reply.started":"2022-08-10T19:32:10.332872Z","shell.execute_reply":"2022-08-10T19:32:10.332922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"it might be useful to look at the average values of taxi fare grouped by each of the predictors.","metadata":{}},{"cell_type":"code","source":"def value_means(df,col):\n    a = df.groupby([col]).mean()['fare_amount']\n    df_value_means = pd.DataFrame(a)\n    df_value_means = df_value_means.reset_index()\n    df_value_means.columns = [f'{col}_unique_values', 'fare_means']\n    return df_value_means","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.336739Z","iopub.status.idle":"2022-08-10T19:32:10.337357Z","shell.execute_reply.started":"2022-08-10T19:32:10.337052Z","shell.execute_reply":"2022-08-10T19:32:10.337080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import display","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.338679Z","iopub.status.idle":"2022-08-10T19:32:10.339265Z","shell.execute_reply.started":"2022-08-10T19:32:10.338974Z","shell.execute_reply":"2022-08-10T19:32:10.339002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in predictors:\n    display(value_means(train,i).applymap('{:,.2f}'.format))","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.341752Z","iopub.status.idle":"2022-08-10T19:32:10.342177Z","shell.execute_reply.started":"2022-08-10T19:32:10.341982Z","shell.execute_reply":"2022-08-10T19:32:10.342002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in predictors:\n    #display(value_means(train,i))\n    sns.scatterplot(x = value_means(train,i).iloc[:,0], y = value_means(train,i).iloc[:,1])\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.343657Z","iopub.status.idle":"2022-08-10T19:32:10.344115Z","shell.execute_reply.started":"2022-08-10T19:32:10.343877Z","shell.execute_reply":"2022-08-10T19:32:10.343928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looking at both graphs and tables, as expected, the fare does seem to increase with years. On weekends, taxi fare seems to be higher but the difference does not seem to be of much significance and could be because of the noise in the data. fare seems to be the lowest in January and the highest in october, maybe not much higher than in other months. As for hours, fair is the highest at nighttime, from 4 to 6 o'clock. It could be useful to introduce dummy variables for these night hours.","metadata":{}},{"cell_type":"code","source":"train['night'] = (train['hour'].isin([4,5])).astype('int64')\ntest['night'] = (test['hour'].isin([4,5])).astype('int64')","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.345326Z","iopub.status.idle":"2022-08-10T19:32:10.345724Z","shell.execute_reply.started":"2022-08-10T19:32:10.345532Z","shell.execute_reply":"2022-08-10T19:32:10.345551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictors = predictors + ['night'] + locs + distances","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.347671Z","iopub.status.idle":"2022-08-10T19:32:10.348136Z","shell.execute_reply.started":"2022-08-10T19:32:10.347928Z","shell.execute_reply":"2022-08-10T19:32:10.347950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split    ","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.350606Z","iopub.status.idle":"2022-08-10T19:32:10.351047Z","shell.execute_reply.started":"2022-08-10T19:32:10.350811Z","shell.execute_reply":"2022-08-10T19:32:10.350830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = train['fare_amount']\nX = train[predictors]\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.25,\n                                                    random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.352637Z","iopub.status.idle":"2022-08-10T19:32:10.353092Z","shell.execute_reply.started":"2022-08-10T19:32:10.352860Z","shell.execute_reply":"2022-08-10T19:32:10.352880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras import layers, callbacks\nfrom sklearn.preprocessing import MinMaxScaler","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.354349Z","iopub.status.idle":"2022-08-10T19:32:10.354785Z","shell.execute_reply.started":"2022-08-10T19:32:10.354576Z","shell.execute_reply":"2022-08-10T19:32:10.354597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_sub = test[predictors]\nX_key = test['key']","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.355824Z","iopub.status.idle":"2022-08-10T19:32:10.356308Z","shell.execute_reply.started":"2022-08-10T19:32:10.356085Z","shell.execute_reply":"2022-08-10T19:32:10.356106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaler = MinMaxScaler()\nX_train = scaler.fit_transform(X_train)\nX_test = scaler.transform(X_test)\nX_sub = scaler.transform(X_sub)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.357469Z","iopub.status.idle":"2022-08-10T19:32:10.357870Z","shell.execute_reply.started":"2022-08-10T19:32:10.357671Z","shell.execute_reply":"2022-08-10T19:32:10.357690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"early_stopping = callbacks.EarlyStopping(\n    min_delta=0.001, # minimium amount of change to count as an improvement\n    patience=20, # how many epochs to wait before stopping\n    restore_best_weights=True,\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.358934Z","iopub.status.idle":"2022-08-10T19:32:10.359317Z","shell.execute_reply.started":"2022-08-10T19:32:10.359126Z","shell.execute_reply":"2022-08-10T19:32:10.359144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = keras.Sequential([\n    layers.Dense(256, activation='relu', input_shape=[17]),\n    layers.BatchNormalization(),\n    layers.Dense(128, activation='relu'),\n    layers.BatchNormalization(),\n    layers.Dense(64, activation='relu'),\n    layers.BatchNormalization(),\n    layers.Dense(32, activation='relu'),\n    layers.BatchNormalization(),\n    layers.Dense(8, activation='relu'),\n    layers.BatchNormalization(),\n    layers.Dense(1),\n    \n])\n\nmodel.compile(\n    optimizer='adam',\n    loss='mae',\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.360679Z","iopub.status.idle":"2022-08-10T19:32:10.361082Z","shell.execute_reply.started":"2022-08-10T19:32:10.360876Z","shell.execute_reply":"2022-08-10T19:32:10.360912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    X_train, y_train,\n    validation_data=(X_test, y_test),\n    batch_size=256,\n    epochs=100,\n    callbacks=[early_stopping],\n    verbose=0,\n)\n\nhistory_df = pd.DataFrame(history.history)\nhistory_df.loc[:, ['loss', 'val_loss']].plot();\nprint(\"Minimum validation loss: {}\".format(history_df['val_loss'].min()))","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.362707Z","iopub.status.idle":"2022-08-10T19:32:10.363157Z","shell.execute_reply.started":"2022-08-10T19:32:10.362960Z","shell.execute_reply":"2022-08-10T19:32:10.362981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(X_sub, batch_size=128, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.364508Z","iopub.status.idle":"2022-08-10T19:32:10.364924Z","shell.execute_reply.started":"2022-08-10T19:32:10.364693Z","shell.execute_reply":"2022-08-10T19:32:10.364711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred.ravel()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.366597Z","iopub.status.idle":"2022-08-10T19:32:10.367185Z","shell.execute_reply.started":"2022-08-10T19:32:10.366791Z","shell.execute_reply":"2022-08-10T19:32:10.366809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.DataFrame({'Key':X_key,'fare_amount':y_pred.ravel()})","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.368045Z","iopub.status.idle":"2022-08-10T19:32:10.368424Z","shell.execute_reply.started":"2022-08-10T19:32:10.368244Z","shell.execute_reply":"2022-08-10T19:32:10.368261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T19:32:10.369702Z","iopub.status.idle":"2022-08-10T19:32:10.370121Z","shell.execute_reply.started":"2022-08-10T19:32:10.369910Z","shell.execute_reply":"2022-08-10T19:32:10.369935Z"},"trusted":true},"execution_count":null,"outputs":[]}]}