{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-13T12:35:14.083490Z","iopub.execute_input":"2022-08-13T12:35:14.084402Z","iopub.status.idle":"2022-08-13T12:35:14.102758Z","shell.execute_reply.started":"2022-08-13T12:35:14.084279Z","shell.execute_reply":"2022-08-13T12:35:14.101779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n%matplotlib inline\nfrom mpl_toolkits import mplot3d\nimport seaborn as sns\n\nimport math\nfrom math import sqrt\n\nfrom numpy import absolute\nfrom numpy import mean\nfrom numpy import std\n\nfrom sklearn import metrics\nfrom sklearn.feature_selection import f_regression\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.preprocessing import PolynomialFeatures\nfrom sklearn.preprocessing import scale\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error, r2_score\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import KFold \nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.linear_model import Ridge\nfrom sklearn.linear_model import RidgeCV\nfrom sklearn.model_selection import RepeatedKFold\nfrom sklearn import neighbors\n\n\nimport tensorflow as tf\nfrom tensorflow.keras import Model\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.optimizers import Adam\nfrom sklearn.preprocessing import StandardScaler\nfrom tensorflow.keras.layers import Dense, Dropout\nfrom tensorflow.keras.losses import MeanSquaredError \nfrom keras.layers import BatchNormalization\nfrom keras.models import Sequential\nfrom keras.layers import Dense\nfrom keras.layers import Dropout\nfrom tensorflow.keras.optimizers import Adam\nfrom keras.callbacks import EarlyStopping\nfrom tensorflow.random import set_seed\n\nimport xgboost as xgb\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:35:14.108620Z","iopub.execute_input":"2022-08-13T12:35:14.108941Z","iopub.status.idle":"2022-08-13T12:35:16.652154Z","shell.execute_reply.started":"2022-08-13T12:35:14.108911Z","shell.execute_reply":"2022-08-13T12:35:16.650960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### There is a lot of data, which will require a lot of computing resources, so I took only a part of it","metadata":{}},{"cell_type":"code","source":"#pickup_datetime was represented as an object so I converted it to a time variable\ntrain = pd.read_csv(\"../input/new-york-city-taxi-fare-prediction/train.csv\", nrows = 500000, parse_dates=[\"pickup_datetime\"]\n                   )\ntest = pd.read_csv(\"../input/new-york-city-taxi-fare-prediction/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:35:16.653590Z","iopub.execute_input":"2022-08-13T12:35:16.654470Z","iopub.status.idle":"2022-08-13T12:36:32.552152Z","shell.execute_reply.started":"2022-08-13T12:35:16.654423Z","shell.execute_reply":"2022-08-13T12:36:32.550744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.shape)\nprint(test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:36:32.557296Z","iopub.execute_input":"2022-08-13T12:36:32.557811Z","iopub.status.idle":"2022-08-13T12:36:32.564822Z","shell.execute_reply.started":"2022-08-13T12:36:32.557764Z","shell.execute_reply":"2022-08-13T12:36:32.563510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:36:32.566839Z","iopub.execute_input":"2022-08-13T12:36:32.567290Z","iopub.status.idle":"2022-08-13T12:36:32.602938Z","shell.execute_reply.started":"2022-08-13T12:36:32.567246Z","shell.execute_reply":"2022-08-13T12:36:32.601625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:36:32.604726Z","iopub.execute_input":"2022-08-13T12:36:32.605180Z","iopub.status.idle":"2022-08-13T12:36:32.615213Z","shell.execute_reply.started":"2022-08-13T12:36:32.605130Z","shell.execute_reply":"2022-08-13T12:36:32.614232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:36:32.616555Z","iopub.execute_input":"2022-08-13T12:36:32.617747Z","iopub.status.idle":"2022-08-13T12:36:32.865666Z","shell.execute_reply.started":"2022-08-13T12:36:32.617713Z","shell.execute_reply":"2022-08-13T12:36:32.864530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## From this we learn that\n* The minimum fare is negative, which is impossible\n* Some travel points are missing the city\n* The maximum number of passengers is equal to 208, which is impossible\n* The maximum fare also unreal","metadata":{}},{"cell_type":"markdown","source":"# Data Cleaning & Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"#### Find, if we have data that has no value","metadata":{}},{"cell_type":"code","source":"print(train.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:36:32.867070Z","iopub.execute_input":"2022-08-13T12:36:32.867449Z","iopub.status.idle":"2022-08-13T12:36:32.906146Z","shell.execute_reply.started":"2022-08-13T12:36:32.867416Z","shell.execute_reply":"2022-08-13T12:36:32.904735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.dropna(how = 'any', axis = 'rows')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:36:32.907632Z","iopub.execute_input":"2022-08-13T12:36:32.908014Z","iopub.status.idle":"2022-08-13T12:36:32.997244Z","shell.execute_reply.started":"2022-08-13T12:36:32.907981Z","shell.execute_reply":"2022-08-13T12:36:32.996070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Old size: %d' % len(train))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:36:32.998631Z","iopub.execute_input":"2022-08-13T12:36:32.998979Z","iopub.status.idle":"2022-08-13T12:36:33.007711Z","shell.execute_reply.started":"2022-08-13T12:36:32.998947Z","shell.execute_reply":"2022-08-13T12:36:33.006241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### There will be no negative tax and you may not be able to pay more than a certain limit depending on the circumstances, let's say this limit is 200\\\\$ \nAlso, the minimum fare for a New York taxi is 2,50\\\\$  ","metadata":{}},{"cell_type":"code","source":"train = train.drop(train[train.fare_amount<2.5].index, axis = 0)\ntrain = train.drop(train[train.fare_amount>300].index, axis = 0)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:36:33.010517Z","iopub.execute_input":"2022-08-13T12:36:33.010966Z","iopub.status.idle":"2022-08-13T12:36:33.164495Z","shell.execute_reply.started":"2022-08-13T12:36:33.010921Z","shell.execute_reply":"2022-08-13T12:36:33.163324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Delete the data whose passenger_count exceeded 6, because it cannot physically fit more in the taxi and it is not allowed. Taxi can also move without passenger and carry cargo, so lets permit passenger_count == 0 data\n","metadata":{}},{"cell_type":"code","source":"train = train.drop(train[train['passenger_count']>6].index, axis = 0)\ntrain = train.drop(train[train['passenger_count']<0].index, axis = 0)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:36:33.165837Z","iopub.execute_input":"2022-08-13T12:36:33.166169Z","iopub.status.idle":"2022-08-13T12:36:33.279087Z","shell.execute_reply.started":"2022-08-13T12:36:33.166140Z","shell.execute_reply":"2022-08-13T12:36:33.277805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Remove the pickup_latitude data whose values are greater than 90 and less than -90, because latitudes are between -90 and 90 degrees","metadata":{}},{"cell_type":"code","source":"train = train.drop(train[train['pickup_latitude']<-90].index, axis = 0)\ntrain = train.drop(train[train['pickup_latitude']>90].index, axis = 0)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:36:33.284656Z","iopub.execute_input":"2022-08-13T12:36:33.285103Z","iopub.status.idle":"2022-08-13T12:36:33.403263Z","shell.execute_reply.started":"2022-08-13T12:36:33.285066Z","shell.execute_reply":"2022-08-13T12:36:33.401678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Remove the pickup_longitude data whose values are greater than 180 and less than -180, because longitudes are between -180 and 180 degrees","metadata":{}},{"cell_type":"code","source":"train = train.drop(train[train['pickup_longitude']<-180].index, axis = 0)\ntrain = train.drop(train[train['pickup_longitude']>180].index, axis = 0)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:36:33.404747Z","iopub.execute_input":"2022-08-13T12:36:33.405786Z","iopub.status.idle":"2022-08-13T12:36:33.531796Z","shell.execute_reply.started":"2022-08-13T12:36:33.405739Z","shell.execute_reply":"2022-08-13T12:36:33.530755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Repeat the same for dropoff_latitude and dropoff_longitude","metadata":{}},{"cell_type":"code","source":"train = train.drop(train[train['dropoff_latitude']<-90].index, axis = 0)\ntrain = train.drop(train[train['dropoff_latitude']>90].index, axis = 0)\n\ntrain = train.drop(train[train['dropoff_longitude']<-180].index, axis = 0)\ntrain = train.drop(train[train['dropoff_longitude']>180].index, axis = 0)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:36:33.532952Z","iopub.execute_input":"2022-08-13T12:36:33.533740Z","iopub.status.idle":"2022-08-13T12:36:33.740690Z","shell.execute_reply.started":"2022-08-13T12:36:33.533705Z","shell.execute_reply":"2022-08-13T12:36:33.739533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### The geographical location may correspond to the real one, but not to New York area, so let's filter the existing data. For this, let's introduce the conditional city limits","metadata":{}},{"cell_type":"code","source":"def select_outside_boundingbox(df, BB):\n    filter_df = df.loc[(df['pickup_longitude'] < BB[0]) | (df['pickup_longitude'] > BB[1]) | \\\n           (df['pickup_latitude'] < BB[2]) | (df['pickup_latitude'] > BB[3]) | \\\n           (df['dropoff_longitude'] < BB[0]) | (df['dropoff_longitude'] > BB[1]) | \\\n           (df['dropoff_latitude'] < BB[2]) | (df['dropoff_latitude'] > BB[3])]\n    \n    return filter_df\n\nNYC_BB = (-74.5, -72.8, 40.5, 41.8)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:36:33.742191Z","iopub.execute_input":"2022-08-13T12:36:33.742551Z","iopub.status.idle":"2022-08-13T12:36:33.750166Z","shell.execute_reply.started":"2022-08-13T12:36:33.742521Z","shell.execute_reply":"2022-08-13T12:36:33.749029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"outliers = select_outside_boundingbox(train, NYC_BB)\noutliers","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:36:33.751777Z","iopub.execute_input":"2022-08-13T12:36:33.752119Z","iopub.status.idle":"2022-08-13T12:36:33.787658Z","shell.execute_reply.started":"2022-08-13T12:36:33.752086Z","shell.execute_reply":"2022-08-13T12:36:33.786773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.drop(outliers.index, axis = 0)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:36:33.788926Z","iopub.execute_input":"2022-08-13T12:36:33.789467Z","iopub.status.idle":"2022-08-13T12:36:33.840392Z","shell.execute_reply.started":"2022-08-13T12:36:33.789434Z","shell.execute_reply":"2022-08-13T12:36:33.839451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('New size: %d' % len(train))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:36:33.841733Z","iopub.execute_input":"2022-08-13T12:36:33.842314Z","iopub.status.idle":"2022-08-13T12:36:33.848241Z","shell.execute_reply.started":"2022-08-13T12:36:33.842266Z","shell.execute_reply":"2022-08-13T12:36:33.846824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#key column and pickup_datetime column of Test set should also be presented as time data\n#test['key'] = pd.to_datetime(test['key'])\n#test['pickup_datetime']  = pd.to_datetime(test['pickup_datetime'])","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:36:33.849772Z","iopub.execute_input":"2022-08-13T12:36:33.850167Z","iopub.status.idle":"2022-08-13T12:36:33.861234Z","shell.execute_reply.started":"2022-08-13T12:36:33.850132Z","shell.execute_reply":"2022-08-13T12:36:33.860171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:36:33.862849Z","iopub.execute_input":"2022-08-13T12:36:33.863482Z","iopub.status.idle":"2022-08-13T12:36:33.877282Z","shell.execute_reply.started":"2022-08-13T12:36:33.863435Z","shell.execute_reply":"2022-08-13T12:36:33.876022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### We can understand displacement through start and end points.\n\n#### We will use the Haversine formula to calculate the distance between two geolocations","metadata":{}},{"cell_type":"code","source":"#საწყისი და საბოლოო კოორდინატების წარმოვადგინე tuple სახით\ntrain[\"loc1\"] = train[[\"pickup_latitude\",\"pickup_longitude\"]].apply(tuple, axis=1)\ntrain[\"loc2\"] = train[[\"dropoff_latitude\",\"dropoff_longitude\"]].apply(tuple, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:36:33.878823Z","iopub.execute_input":"2022-08-13T12:36:33.879453Z","iopub.status.idle":"2022-08-13T12:36:44.306833Z","shell.execute_reply.started":"2022-08-13T12:36:33.879410Z","shell.execute_reply":"2022-08-13T12:36:44.305506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import haversine as hs\n\n        \ntrain['H_Distance'] = train.apply(lambda row: hs.haversine(row.loc1,row.loc2), axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:36:44.309165Z","iopub.execute_input":"2022-08-13T12:36:44.310496Z","iopub.status.idle":"2022-08-13T12:37:00.225241Z","shell.execute_reply.started":"2022-08-13T12:36:44.310453Z","shell.execute_reply":"2022-08-13T12:37:00.223937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Let's also calculate the distance using the Chebyshev method","metadata":{}},{"cell_type":"code","source":"def chebyshev(pickup_long, dropoff_long, pickup_lat, dropoff_lat):\n    return np.maximum(np.absolute(pickup_long - dropoff_long), np.absolute(pickup_lat - dropoff_lat))\n\ntrain['Chebyshev'] = chebyshev(train['pickup_longitude'], train['dropoff_longitude'], train['pickup_latitude'], train['dropoff_latitude'])","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:37:00.226708Z","iopub.execute_input":"2022-08-13T12:37:00.227071Z","iopub.status.idle":"2022-08-13T12:37:00.246237Z","shell.execute_reply.started":"2022-08-13T12:37:00.227039Z","shell.execute_reply":"2022-08-13T12:37:00.244977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:37:00.248042Z","iopub.execute_input":"2022-08-13T12:37:00.249013Z","iopub.status.idle":"2022-08-13T12:37:00.279330Z","shell.execute_reply.started":"2022-08-13T12:37:00.248966Z","shell.execute_reply":"2022-08-13T12:37:00.277728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"hour\"] = train.pickup_datetime.dt.hour\ntrain[\"day_of_week\"] = train.pickup_datetime.dt.weekday\ntrain[\"day_of_month\"] = train.pickup_datetime.dt.day\ntrain[\"week\"] = train.pickup_datetime.dt.week\ntrain[\"month\"] = train.pickup_datetime.dt.month\ntrain[\"year\"] = train.pickup_datetime.dt.year - 2000\n\ntrain['minute'] =train['pickup_datetime'].dt.minute\ntrain['second'] = train['pickup_datetime'].dt.second\ntrain['dayofyear'] = train['pickup_datetime'].dt.dayofyear","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:37:00.281559Z","iopub.execute_input":"2022-08-13T12:37:00.282044Z","iopub.status.idle":"2022-08-13T12:37:00.912429Z","shell.execute_reply.started":"2022-08-13T12:37:00.281997Z","shell.execute_reply":"2022-08-13T12:37:00.911329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:37:00.913735Z","iopub.execute_input":"2022-08-13T12:37:00.914091Z","iopub.status.idle":"2022-08-13T12:37:00.945720Z","shell.execute_reply.started":"2022-08-13T12:37:00.914060Z","shell.execute_reply":"2022-08-13T12:37:00.944622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Add a variable that determines how much each kilometer of travel costs","metadata":{}},{"cell_type":"code","source":"train[\"fare_to_dist_ratio\"] = train[\"fare_amount\"] / ( train[\"H_Distance\"]+0.0001)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:37:00.947332Z","iopub.execute_input":"2022-08-13T12:37:00.947976Z","iopub.status.idle":"2022-08-13T12:37:00.956651Z","shell.execute_reply.started":"2022-08-13T12:37:00.947940Z","shell.execute_reply":"2022-08-13T12:37:00.955504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Remove those whose start and end points match","metadata":{}},{"cell_type":"code","source":"train = train.drop(train[train['loc1']==train['loc2']].index, axis = 0)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:37:00.958499Z","iopub.execute_input":"2022-08-13T12:37:00.959068Z","iopub.status.idle":"2022-08-13T12:37:01.199979Z","shell.execute_reply.started":"2022-08-13T12:37:00.959036Z","shell.execute_reply":"2022-08-13T12:37:01.198742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### It is necessary to take into account the circumstances of calling at the airport, according to which the fee changes dramatically. For this, let's write the given function that shows how many kilometers away the point is from the airport\n","metadata":{}},{"cell_type":"code","source":"def add_distances_from_airport(dataset):\n    #coordinates of all these airports\n    jfk_coords = (40.639722, -73.778889)\n    ewr_coords = (40.6925, -74.168611)\n    lga_coords = (40.77725, -73.872611)\n\n    dataset['pickup_jfk_distance'] = dataset.apply(lambda row: hs.haversine(jfk_coords,row.loc1), axis=1)\n    dataset['dropof_jfk_distance'] = dataset.apply(lambda row: hs.haversine(jfk_coords,row.loc2), axis=1)\n    \n    \n    dataset['pickup_ewr_distance'] = dataset.apply(lambda row: hs.haversine(ewr_coords,row.loc1), axis=1)\n    dataset['dropof_ewr_distance'] = dataset.apply(lambda row: hs.haversine(ewr_coords,row.loc2), axis=1)\n    \n    \n    dataset['pickup_lga_distance'] = dataset.apply(lambda row: hs.haversine(lga_coords,row.loc1), axis=1)\n    dataset['dropof_lga_distance'] = dataset.apply(lambda row: hs.haversine(lga_coords,row.loc2), axis=1)\n\n    return dataset\n\n\ntrain = add_distances_from_airport(train)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:37:01.201483Z","iopub.execute_input":"2022-08-13T12:37:01.201852Z","iopub.status.idle":"2022-08-13T12:38:13.477132Z","shell.execute_reply.started":"2022-08-13T12:37:01.201820Z","shell.execute_reply":"2022-08-13T12:38:13.476071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's repeat the same procedures on the test part","metadata":{}},{"cell_type":"code","source":"test['pickup_datetime']  = pd.to_datetime(test['pickup_datetime'])\n\ntest[\"loc1\"] = test[[\"pickup_latitude\",\"pickup_longitude\"]].apply(tuple, axis=1)\ntest[\"loc2\"] = test[[\"dropoff_latitude\",\"dropoff_longitude\"]].apply(tuple, axis=1)\n\ntest['H_Distance'] = test.apply(lambda row: hs.haversine(row.loc1,row.loc2), axis=1)\n\ntest['Chebyshev'] = chebyshev(test['pickup_longitude'], test['dropoff_longitude'], test['pickup_latitude'], test['dropoff_latitude'])\n\n\ntest[\"hour\"] = test.pickup_datetime.dt.hour\ntest[\"day_of_week\"] = test.pickup_datetime.dt.weekday\ntest[\"day_of_month\"] = test.pickup_datetime.dt.day\ntest[\"week\"] = test.pickup_datetime.dt.week\ntest[\"month\"] = test.pickup_datetime.dt.month\ntest[\"year\"] = test.pickup_datetime.dt.year - 2000\n\ntest['minute'] =test['pickup_datetime'].dt.minute\ntest['second'] = test['pickup_datetime'].dt.second\ntest['dayofyear'] = test['pickup_datetime'].dt.dayofyear\n\ntest = add_distances_from_airport(test)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:38:13.478600Z","iopub.execute_input":"2022-08-13T12:38:13.479026Z","iopub.status.idle":"2022-08-13T12:38:15.738275Z","shell.execute_reply.started":"2022-08-13T12:38:13.478995Z","shell.execute_reply":"2022-08-13T12:38:15.737068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#დიდ მონაცემთა ბაზას ჭირდება დიდი გამოთვლითი რესურსი, მოცემული ფუნქციით, რიცხვით ტიპებს დავიყვანთ შესაძლო მინიმალურ მოცულობამდე,რაც გაამარტივებს გამოთვლებს\ndef downcast(df):\n\n    df_int = df.select_dtypes(include=['int64', 'int32', 'int16', 'int8', 'int'])\n    df[df_int.columns] = df_int.apply(pd.to_numeric,downcast='unsigned')\n    \n    df_float = df.select_dtypes(include=['float64', 'float32', 'float16', 'float'])\n    df[df_float.columns] = df_float.apply(pd.to_numeric,downcast='float')\n        \n    return df\ndowncast(train)\ndowncast(test)\ntrain.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:38:15.739870Z","iopub.execute_input":"2022-08-13T12:38:15.740222Z","iopub.status.idle":"2022-08-13T12:38:16.113262Z","shell.execute_reply.started":"2022-08-13T12:38:15.740190Z","shell.execute_reply":"2022-08-13T12:38:16.112169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Map visualization\n","metadata":{}},{"cell_type":"code","source":"train.plot(y='pickup_latitude',x='pickup_longitude',kind=\"scatter\",alpha=0.7,s=0.02)\ncity_long_border = (-74.03, -73.75)\ncity_lat_border = (40.63, 40.85)\nplt.title(\"Pickups Data\")\n\nplt.ylim(city_lat_border)\nplt.xlim(city_long_border)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:38:16.114759Z","iopub.execute_input":"2022-08-13T12:38:16.115079Z","iopub.status.idle":"2022-08-13T12:38:19.025337Z","shell.execute_reply.started":"2022-08-13T12:38:16.115051Z","shell.execute_reply":"2022-08-13T12:38:19.024068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.plot(y='dropoff_latitude',x='dropoff_longitude',kind=\"scatter\",alpha=0.5,s=0.02)\ncity_long_border = (-74.03, -73.75)\ncity_lat_border = (40.63, 40.85)\nplt.title(\"Dropoff Data\")\n\nplt.ylim(city_lat_border)\nplt.xlim(city_long_border)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:38:19.026898Z","iopub.execute_input":"2022-08-13T12:38:19.027272Z","iopub.status.idle":"2022-08-13T12:38:21.426604Z","shell.execute_reply.started":"2022-08-13T12:38:19.027236Z","shell.execute_reply":"2022-08-13T12:38:21.425369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import folium\n\n\nlong_trips=train[train['H_Distance']>=10]\n\ndrop_map = folium.Map(location = [40.730610,-73.935242],zoom_start = 12)\n\n### For each pickup point add a circlemarker\n\nfor index, row in long_trips.iterrows():\n    \n    folium.CircleMarker([row['dropoff_latitude'], row['dropoff_longitude']],\n                        radius=3,\n                        color=\"green\", \n                        fill_opacity=0.9\n                       ).add_to(drop_map)\nfor index, row in long_trips.iterrows():\n    \n    folium.CircleMarker([row['pickup_latitude'], row['pickup_longitude']],\n                        radius=3,\n                        color=\"blue\", \n                        fill_opacity=0.9\n                       ).add_to(drop_map)\ndrop_map","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:38:21.428507Z","iopub.execute_input":"2022-08-13T12:38:21.428866Z","iopub.status.idle":"2022-08-13T12:38:59.399994Z","shell.execute_reply.started":"2022-08-13T12:38:21.428834Z","shell.execute_reply":"2022-08-13T12:38:59.397836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## EDA","metadata":{}},{"cell_type":"code","source":"sns.histplot(data=train,x=\"fare_amount\",kde=True)\n\nplt.axvline(train[\"fare_amount\"].mean(),color = \"k\",\n            linestyle = \"dashed\",label = \"Avg trip fare ($)\")\nplt.axvline(train[\"fare_amount\"].median(),color = \"r\",\n            linestyle = \"dashed\",label = \"Median trip fare ($)\")\n\nplt.title(\"Distribution of fare amount\")\nplt.xticks(np.arange(0, 100, step=5))\nplt.legend(loc = \"best\",prop = {\"size\" : 12})\nplt.gcf().set_size_inches(15,8)\nplt.xlim(0,100)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:38:59.402705Z","iopub.execute_input":"2022-08-13T12:38:59.403528Z","iopub.status.idle":"2022-08-13T12:39:05.885833Z","shell.execute_reply.started":"2022-08-13T12:38:59.403440Z","shell.execute_reply":"2022-08-13T12:39:05.884614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### A right-skewed distribution\n* #### Most taxi fares range from $2.5\\\\$-20\\\\$.\n* #### The average taxi fee varies between 10\\\\$-12\\\\$\n* #### 45\\\\$-50\\\\$-57\\\\$ peaks are observed, which fixed fee","metadata":{}},{"cell_type":"code","source":"sns.histplot(data=train,x=\"H_Distance\",kde=True)\n\nplt.axvline(train[\"H_Distance\"].mean(),color = \"k\",\n            linestyle = \"dashed\",label = \"Avg H_Distance (km)\")\nplt.axvline(train[\"H_Distance\"].median(),color = \"r\",\n            linestyle = \"dashed\",label = \"Median H_Distance (km)\")\n\nplt.title(\"Distribution of fare amount\")\nplt.xticks(np.arange(0, 30, step=5))\nplt.legend(loc = \"best\",prop = {\"size\" : 12})\nplt.gcf().set_size_inches(15,8)\nplt.xlim(0,30)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:05.887701Z","iopub.execute_input":"2022-08-13T12:39:05.888436Z","iopub.status.idle":"2022-08-13T12:39:11.337144Z","shell.execute_reply.started":"2022-08-13T12:39:05.888399Z","shell.execute_reply":"2022-08-13T12:39:11.336261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* #### Passengers travel an average of 3-5 km by taxi","metadata":{}},{"cell_type":"markdown","source":"#### 1. Does the number of passengers affect the fare?","metadata":{}},{"cell_type":"code","source":"sns.set(rc = {'figure.figsize':(15,8)})\nsns.histplot(data=train, x=\"passenger_count\", stat=\"count\", discrete=True)\nplt.title(\"Distribution of passenger count\")\nplt.xlabel('No. of Passengers')\nplt.ylabel('Frequency')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:11.338660Z","iopub.execute_input":"2022-08-13T12:39:11.339277Z","iopub.status.idle":"2022-08-13T12:39:11.703719Z","shell.execute_reply.started":"2022-08-13T12:39:11.339229Z","shell.execute_reply":"2022-08-13T12:39:11.702544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* #### Most of the passengers were traveling alone","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15,8))\nplt.scatter(x=train['passenger_count'], y=train['fare_amount'], s=1.5)\nplt.xlabel('No. of Passengers')\nplt.ylabel('Fare')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:11.705100Z","iopub.execute_input":"2022-08-13T12:39:11.705514Z","iopub.status.idle":"2022-08-13T12:39:12.255703Z","shell.execute_reply.started":"2022-08-13T12:39:11.705476Z","shell.execute_reply":"2022-08-13T12:39:12.254676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* #### One-passenger taxi has more passengers whose fare is higher","metadata":{}},{"cell_type":"markdown","source":"#### 2. Does the pick-up date and time affect the fare?","metadata":{}},{"cell_type":"code","source":"colormap = plt.cm.RdBu\nplt.figure(figsize=(20, 20))\nheatmap = sns.heatmap(train.corr(),linewidths=0.1,vmax=1.0,vmin=0, \n            square=True, cmap=\"Blues\", linecolor='white', annot=True)\nheatmap.set_title('Correlation Heatmap', fontdict={'fontsize':12}, pad=12);","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:12.263811Z","iopub.execute_input":"2022-08-13T12:39:12.264192Z","iopub.status.idle":"2022-08-13T12:39:15.992919Z","shell.execute_reply.started":"2022-08-13T12:39:12.264155Z","shell.execute_reply":"2022-08-13T12:39:15.991337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,8))\nplt.scatter(x=train['year'], y=train['fare_amount'], s=1.5)\nplt.xlabel('Years')\nplt.ylabel('Fare')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:15.994758Z","iopub.execute_input":"2022-08-13T12:39:15.995450Z","iopub.status.idle":"2022-08-13T12:39:16.542338Z","shell.execute_reply.started":"2022-08-13T12:39:15.995407Z","shell.execute_reply":"2022-08-13T12:39:16.541055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[['fare_amount','year']].corr()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:16.543882Z","iopub.execute_input":"2022-08-13T12:39:16.545930Z","iopub.status.idle":"2022-08-13T12:39:16.575837Z","shell.execute_reply.started":"2022-08-13T12:39:16.545883Z","shell.execute_reply":"2022-08-13T12:39:16.574567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* #### The rate does not change significantly over the years","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15,8))\nplt.scatter(x=train['month'], y=train['fare_amount'], s=1.5)\nplt.xlabel('Month')\nplt.ylabel('Fare')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:16.577667Z","iopub.execute_input":"2022-08-13T12:39:16.578437Z","iopub.status.idle":"2022-08-13T12:39:17.124630Z","shell.execute_reply.started":"2022-08-13T12:39:16.578389Z","shell.execute_reply":"2022-08-13T12:39:17.123393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* #### The rate is uniform throughout the months","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15,8))\nplt.scatter(x=train['day_of_month'], y=train['fare_amount'], s=1.5)\nplt.xlabel('Days')\nplt.ylabel('Fare')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:17.126279Z","iopub.execute_input":"2022-08-13T12:39:17.127506Z","iopub.status.idle":"2022-08-13T12:39:17.678418Z","shell.execute_reply.started":"2022-08-13T12:39:17.127457Z","shell.execute_reply":"2022-08-13T12:39:17.677321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* #### The fare is uniform throughout the month","metadata":{}},{"cell_type":"code","source":"sns.set(rc = {'figure.figsize':(15,8)})\nsns.histplot(data=train, x=\"hour\", stat=\"count\", discrete=True, kde=True)\nplt.title(\"Distribution by Hours\")\nplt.xlabel('Hours')\nplt.ylabel('Frequency')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:17.680269Z","iopub.execute_input":"2022-08-13T12:39:17.680824Z","iopub.status.idle":"2022-08-13T12:39:19.713383Z","shell.execute_reply.started":"2022-08-13T12:39:17.680786Z","shell.execute_reply":"2022-08-13T12:39:19.712090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* #### Taxi fares are rare at 5am and reaches the maximum at 7pm","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15,7))\nplt.scatter(x=train['hour'], y=train['fare_amount'], s=1.5)\nplt.xlabel('Hour')\nplt.ylabel('Fare')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:19.714930Z","iopub.execute_input":"2022-08-13T12:39:19.716000Z","iopub.status.idle":"2022-08-13T12:39:20.243876Z","shell.execute_reply.started":"2022-08-13T12:39:19.715955Z","shell.execute_reply":"2022-08-13T12:39:20.242853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"train[train.passenger_count <7][['fare_amount','passenger_count']].corr()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:20.245174Z","iopub.execute_input":"2022-08-13T12:39:20.245546Z","iopub.status.idle":"2022-08-13T12:39:20.384515Z","shell.execute_reply.started":"2022-08-13T12:39:20.245514Z","shell.execute_reply":"2022-08-13T12:39:20.383284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def time_slicer(df, timeframes, value, color=\"purple\"):\n    \"\"\"\n    Function to count observation occurrence through different lenses of time.\n    \"\"\"\n    f, ax = plt.subplots(len(timeframes), figsize = [12,12])\n    for i,x in enumerate(timeframes):\n        df.loc[:,[x,value]].groupby([x]).mean().plot(ax=ax[i],color=color)\n        ax[i].set_ylabel(value.replace(\"_\", \" \").title())\n        ax[i].set_title(\"{} by {}\".format(value.replace(\"_\", \" \").title(), x.replace(\"_\", \" \").title()))\n        ax[i].set_xlabel(\"\")\n    plt.tight_layout(pad=0)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:20.385702Z","iopub.execute_input":"2022-08-13T12:39:20.386026Z","iopub.status.idle":"2022-08-13T12:39:20.394596Z","shell.execute_reply.started":"2022-08-13T12:39:20.385998Z","shell.execute_reply":"2022-08-13T12:39:20.393233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"time_slicer(df=train, timeframes=['hour', 'day_of_week','day_of_month', 'week', 'month', 'year',], value = \"fare_amount\", color=\"blue\")","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:20.396151Z","iopub.execute_input":"2022-08-13T12:39:20.397385Z","iopub.status.idle":"2022-08-13T12:39:21.754128Z","shell.execute_reply.started":"2022-08-13T12:39:20.397335Z","shell.execute_reply":"2022-08-13T12:39:21.750841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* #### The higher the demand, the lower the fee and vice versa\n* #### Average fares peak on Mondays and Thursdays\n* #### The average fee has been increasing over the years ","metadata":{}},{"cell_type":"code","source":"time_slicer(df=train, timeframes=['hour', 'day_of_week','day_of_month', 'week', 'month', 'year',], value = \"H_Distance\", color=\"green\")","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:21.756376Z","iopub.execute_input":"2022-08-13T12:39:21.757120Z","iopub.status.idle":"2022-08-13T12:39:23.127659Z","shell.execute_reply.started":"2022-08-13T12:39:21.757047Z","shell.execute_reply":"2022-08-13T12:39:23.126382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[['fare_amount','H_Distance']].corr()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:23.129322Z","iopub.execute_input":"2022-08-13T12:39:23.129761Z","iopub.status.idle":"2022-08-13T12:39:23.159772Z","shell.execute_reply.started":"2022-08-13T12:39:23.129712Z","shell.execute_reply":"2022-08-13T12:39:23.158717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* #### The correlation number between these two values is high, as a result the graphs are very similar to each other\n* #### At 5 o'clock in the morning, some people who have a long distance to travel leave home early, but they are not many, so the fare increases.","metadata":{}},{"cell_type":"code","source":"time_slicer(df=train, timeframes=['hour', 'day_of_week','day_of_month', 'week', 'month', 'year',], value = \"fare_to_dist_ratio\", color=\"red\")","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:23.161120Z","iopub.execute_input":"2022-08-13T12:39:23.162049Z","iopub.status.idle":"2022-08-13T12:39:24.517662Z","shell.execute_reply.started":"2022-08-13T12:39:23.162014Z","shell.execute_reply":"2022-08-13T12:39:24.516509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* #### The price of 1 km is high at the beginning of the day and decreases during the day\n* #### The price of 1 km increases during the week\n* #### The price of 1 km decreases during the months","metadata":{}},{"cell_type":"code","source":"# scatter plot distance - fare\nfig, axs = plt.subplots(1, 2, figsize=(16,6))\naxs[0].scatter(train.H_Distance, train.fare_amount, alpha=0.2)\naxs[0].set_xlabel('H_Distance mile')\naxs[0].set_ylabel('fare $USD')\naxs[0].set_title('All data')\n\n# zoom in on part of data\nidx = (train.H_Distance < 30) & (train.fare_amount < 100)\naxs[1].scatter(train[idx].H_Distance, train[idx].fare_amount, alpha=0.2)\naxs[1].set_xlabel('H_Distance mile')\naxs[1].set_ylabel('fare $USD')\naxs[1].set_title('Zoom in on distance < 30 mile, fare < $100');","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:24.519247Z","iopub.execute_input":"2022-08-13T12:39:24.519884Z","iopub.status.idle":"2022-08-13T12:39:26.916415Z","shell.execute_reply.started":"2022-08-13T12:39:24.519847Z","shell.execute_reply":"2022-08-13T12:39:26.915235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Average $USD/Km : {:0.2f}\".format(train.fare_amount.sum()/train.H_Distance.sum()))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:26.917834Z","iopub.execute_input":"2022-08-13T12:39:26.918619Z","iopub.status.idle":"2022-08-13T12:39:26.926818Z","shell.execute_reply.started":"2022-08-13T12:39:26.918571Z","shell.execute_reply":"2022-08-13T12:39:26.925524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* #### The horizontal lines on the graph to the right may indicate fixed fares from the airport\n* #### Overall, a linear relationship is observed","metadata":{}},{"cell_type":"code","source":"train.reset_index(drop=True, inplace=True)\n\ntrain[train[\"fare_to_dist_ratio\"]>500]","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:26.928538Z","iopub.execute_input":"2022-08-13T12:39:26.928878Z","iopub.status.idle":"2022-08-13T12:39:26.978173Z","shell.execute_reply.started":"2022-08-13T12:39:26.928840Z","shell.execute_reply":"2022-08-13T12:39:26.976990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"passenger_fare = train.groupby(['passenger_count']).mean()\nsns.barplot(x=passenger_fare.index, y=passenger_fare['fare_amount'], palette = \"Set3\")\nplt.xlabel('Number of Passengers')\nplt.ylabel('Average Fare Price')\nplt.title('Average Fare Price for Number of Passengers')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:26.979702Z","iopub.execute_input":"2022-08-13T12:39:26.980059Z","iopub.status.idle":"2022-08-13T12:39:27.351576Z","shell.execute_reply.started":"2022-08-13T12:39:26.980028Z","shell.execute_reply":"2022-08-13T12:39:27.350636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preprocess","metadata":{}},{"cell_type":"code","source":"train.drop(['key', 'pickup_datetime','loc1','loc2',\"fare_to_dist_ratio\"], axis=1, inplace=True)\n\ntest_key = pd.DataFrame(test['key'])\ntest.drop(['key', 'pickup_datetime','loc1','loc2',], axis=1, inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:27.353039Z","iopub.execute_input":"2022-08-13T12:39:27.353423Z","iopub.status.idle":"2022-08-13T12:39:27.390327Z","shell.execute_reply.started":"2022-08-13T12:39:27.353388Z","shell.execute_reply":"2022-08-13T12:39:27.389134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.dtypes)\nprint(test.dtypes)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:27.391986Z","iopub.execute_input":"2022-08-13T12:39:27.392358Z","iopub.status.idle":"2022-08-13T12:39:27.400531Z","shell.execute_reply.started":"2022-08-13T12:39:27.392318Z","shell.execute_reply":"2022-08-13T12:39:27.399367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Split Data","metadata":{}},{"cell_type":"code","source":"X = train.iloc[:,train.columns != \"fare_amount\"]\ny = train[\"fare_amount\"]\nX_test = test\nX_train, X_valid, y_train, y_valid = train_test_split(X, y, train_size=0.75, test_size=0.25, random_state=42, shuffle=True)\n\n\n# scaler = MinMaxScaler(feature_range=(0, 1))\n\n# X_train_scaled = scaler.fit_transform(X_train)\n# X_train = pd.DataFrame(X_train_scaled)\n\n# X_valid_scaled = scaler.fit_transform(X_valid)\n# X_valid = pd.DataFrame(X_valid_scaled)\n\n# X_test_scaled = scaler.fit_transform(X_test)\n# X_test = pd.DataFrame(X_test_scaled)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:27.402197Z","iopub.execute_input":"2022-08-13T12:39:27.402567Z","iopub.status.idle":"2022-08-13T12:39:27.527596Z","shell.execute_reply.started":"2022-08-13T12:39:27.402536Z","shell.execute_reply":"2022-08-13T12:39:27.526344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model 1 : Linear Regression","metadata":{}},{"cell_type":"code","source":"# def Linear_reg(X_train, X_valid, y_train, y_valid):    \n#     linear = LinearRegression()\n#     linear.fit(X_train, y_train)\n    \n#     y_train_predict = linear.predict(X_train)\n#     r2_train = r2_score(y_train, y_train_predict)\n#     RMSE_train = mean_squared_error(y_train, y_train_predict, squared=False)\n    \n#     y_valid_predict = linear.predict(X_valid)\n#     r2_valid = r2_score(y_valid, y_valid_predict)\n#     RMSE_valid = mean_squared_error(y_valid, y_valid_predict, squared=False)\n    \n#     return r2_train, r2_valid, linear,RMSE_train,RMSE_valid\n\n# r2_train, r2_valid, linear,RMSE_train,RMSE_valid = Linear_reg(X_train, X_valid, y_train, y_valid)\n\n# #print(\"R^2 (train) : \", r2_train)\n# print(\"RMSE (train): \", RMSE_train)\n# #print(\"R^2 (valid) : \", r2_valid)\n# print(\"RMSE (valid): \", RMSE_valid)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:27.529434Z","iopub.execute_input":"2022-08-13T12:39:27.529781Z","iopub.status.idle":"2022-08-13T12:39:27.535262Z","shell.execute_reply.started":"2022-08-13T12:39:27.529749Z","shell.execute_reply":"2022-08-13T12:39:27.534002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* RMSE (train):  1.3202917e-05\n* RMSE (valid):  1.4740558","metadata":{}},{"cell_type":"markdown","source":"## Model 2 : Polynomial Regression\n","metadata":{}},{"cell_type":"code","source":"# def Poly_reg(X_train, X_valid, y_train, y_valid, degree={'degree' : 2}):\n#     poly = PolynomialFeatures(degree=degree['degree'])\n#     X_train_ = poly.fit_transform(X_train)\n#     est_poly = LinearRegression()    \n#     est_poly.fit(X_train_,y_train) \n\n#     y_train_predict = est_poly.predict(X_train_)\n#     r2_train = r2_score(y_train, y_train_predict)\n#     RMSE_train = mean_squared_error(y_train, y_train_predict, squared=False)\n    \n    \n#     X_valid_ = poly.fit_transform(X_valid)\n\n#     y_valid_predict = est_poly.predict(X_valid_)\n#     r2_valid = r2_score(y_valid, y_valid_predict)\n#     RMSE_valid = mean_squared_error(y_valid, y_valid_predict, squared=False)\n\n#     return r2_train, r2_valid, RMSE_train, RMSE_valid\n\n# for deg in range(1,5):\n#     degree={'degree' : deg}\n#     r2_train, r2_valid,RMSE_train, RMSE_valid = Poly_reg(X_train, X_valid, y_train, y_valid, degree)\n#     print(\"Polynomial degree\", deg)\n#     print(\"R^2 (train) : \", r2_train)\n#     print(\"RMSE (train): \", RMSE_train)\n#     print(\"R^2 (valid) : \", r2_valid)\n#     print(\"RMSE (valid): \", RMSE_valid)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:27.536612Z","iopub.execute_input":"2022-08-13T12:39:27.536949Z","iopub.status.idle":"2022-08-13T12:39:27.545990Z","shell.execute_reply.started":"2022-08-13T12:39:27.536918Z","shell.execute_reply":"2022-08-13T12:39:27.544752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Polynomial degree 1\n* R^2 (train) :  0.9999999999953564\n* RMSE (train):  2.0643134e-05\n* R^2 (valid) :  0.3986446655631378\n* RMSE (valid):  7.342692\n* Polynomial degree 2\n* R^2 (train) :  0.9999999998963758\n* RMSE (train):  9.751603e-05\n* R^2 (valid) :  0.40283202490610714\n* RMSE (valid):  7.317083\n* Polynomial degree 3\n* R^2 (train) :  0.999999991415696\n* RMSE (train):  0.00088756054\n* R^2 (valid) :  0.39955006653268765\n* RMSE (valid):  7.3371625","metadata":{}},{"cell_type":"markdown","source":"## Model 2 : Ridge Regression","metadata":{}},{"cell_type":"code","source":"\n# model = Ridge(alpha=1.0)\n# cv = RepeatedKFold(n_splits=10, n_repeats=3, random_state=1)\n# scores = cross_val_score(model, X_train, y_train, scoring='neg_root_mean_squared_error', cv=cv, n_jobs=-1)\n# scores = absolute(scores)\n# print('rmse: %.3f (%.3f)' % (mean(scores), std(scores)))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:27.547541Z","iopub.execute_input":"2022-08-13T12:39:27.547888Z","iopub.status.idle":"2022-08-13T12:39:27.559352Z","shell.execute_reply.started":"2022-08-13T12:39:27.547857Z","shell.execute_reply":"2022-08-13T12:39:27.558469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"rmse: 0.041 (0.003)","metadata":{}},{"cell_type":"markdown","source":"## Model 3: DECISION TREE REGRESSOR MODEL","metadata":{}},{"cell_type":"code","source":"# dtr = DecisionTreeRegressor().fit(X_train, y_train)\n# y_valid_pred = dtr.predict(X_valid)\n\n\n\n# # Root Mean Square Error\n# rmse = np.sqrt(mean_squared_error(y_valid, y_valid_pred))\n# print(\"RMSE: %f\" % (rmse))\n\n# # Mean Squared Error\n# print(\"Mean squared error: %.2f\"\n#       % mean_squared_error(y_valid, y_valid_pred))\n\n# # R2 Score\n# print('Variance score: %.2f' % r2_score(y_valid, y_valid_pred))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:27.560336Z","iopub.execute_input":"2022-08-13T12:39:27.560668Z","iopub.status.idle":"2022-08-13T12:39:27.568869Z","shell.execute_reply.started":"2022-08-13T12:39:27.560638Z","shell.execute_reply":"2022-08-13T12:39:27.567958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* RMSE: 1.474739\n* Mean squared error: 2.17\n* Variance score: 0.98\n* when size = 1000000","metadata":{}},{"cell_type":"markdown","source":"## Model 4: KNR","metadata":{}},{"cell_type":"code","source":"# rmse_val = [] \n# for K in range(20):\n#     K = K+1\n#     model = neighbors.KNeighborsRegressor(n_neighbors = K)\n\n#     model.fit(X_train, y_train)  \n#     pred=model.predict(X_valid) \n#     error = sqrt(mean_squared_error(y_valid,pred)) \n#     rmse_val.append(error) \n#     print('RMSE value for k= ' , K , 'is:', error)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:27.569793Z","iopub.execute_input":"2022-08-13T12:39:27.570105Z","iopub.status.idle":"2022-08-13T12:39:27.579728Z","shell.execute_reply.started":"2022-08-13T12:39:27.570076Z","shell.execute_reply":"2022-08-13T12:39:27.578867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# curve = pd.DataFrame(rmse_val) \n# curve.plot()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:27.580659Z","iopub.execute_input":"2022-08-13T12:39:27.580972Z","iopub.status.idle":"2022-08-13T12:39:27.594170Z","shell.execute_reply.started":"2022-08-13T12:39:27.580944Z","shell.execute_reply":"2022-08-13T12:39:27.593287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model 5: Support Vector Machines","metadata":{}},{"cell_type":"code","source":"# svr = svm.LinearSVR()\n# svr.fit(X_train,y_train)\n# test_val_pred_svr = svr.predict(X_test_val)\n# np.sqrt(metrics.mean_squared_error(y_test_val,test_val_pred_svr))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:27.595524Z","iopub.execute_input":"2022-08-13T12:39:27.595865Z","iopub.status.idle":"2022-08-13T12:39:27.604069Z","shell.execute_reply.started":"2022-08-13T12:39:27.595833Z","shell.execute_reply":"2022-08-13T12:39:27.602977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model 6: Neural Network","metadata":{}},{"cell_type":"code","source":"# early_stop = EarlyStopping(monitor='val_loss', mode='min', min_delta=1e-3, patience=3)\n# callback=[early_stop]\n# adam = Adam(lr=0.0001)\n\n# model = Sequential() \n# model.add(Dense(100, activation='relu', input_shape=(train.shape[1],)))\n# model.add(Dropout(0.6))\n# model.add(Dense(80, activation='relu'))\n# model.add(Dropout(0.6))\n# model.add(Dense(40, activation='relu'))\n# model.add(Dropout(0.6))\n# model.add(Dense(1, activation='linear'))\n# model.compile(loss='mse', optimizer=adam, metrics=['mae'])\n# history = model.fit(X_train,y_train,batch_size=256, epochs=50, verbose=1, callbacks=callback,\n#          validation_data=(X_valid, y_valid), shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:27.605439Z","iopub.execute_input":"2022-08-13T12:39:27.605769Z","iopub.status.idle":"2022-08-13T12:39:27.615087Z","shell.execute_reply.started":"2022-08-13T12:39:27.605739Z","shell.execute_reply":"2022-08-13T12:39:27.613951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # summarize history for loss using learning curve\n# plt.plot(history.history['loss'])\n# plt.plot(history.history['val_loss'])\n# plt.title('model loss')\n# plt.ylabel('loss')\n# plt.xlabel('epoch')\n# plt.legend(['train', 'test'], loc='upper left')\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:27.616216Z","iopub.execute_input":"2022-08-13T12:39:27.616581Z","iopub.status.idle":"2022-08-13T12:39:27.625449Z","shell.execute_reply.started":"2022-08-13T12:39:27.616550Z","shell.execute_reply":"2022-08-13T12:39:27.624505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_pred = np.array(model.predict(X_valid))\n# print(sqrt(mean_squared_error(y_valid, y_pred)))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:27.626941Z","iopub.execute_input":"2022-08-13T12:39:27.627297Z","iopub.status.idle":"2022-08-13T12:39:27.634898Z","shell.execute_reply.started":"2022-08-13T12:39:27.627264Z","shell.execute_reply":"2022-08-13T12:39:27.634046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model 7: XGBoost","metadata":{}},{"cell_type":"code","source":"params = {\n   \n    'max_depth': 7,\n    'gamma' :0,\n    'eta':.03, \n    'subsample': 1,\n    'colsample_bytree': 0.9, \n    'objective':'reg:linear',\n    'eval_metric':'rmse',\n    'silent': 0,\n    'verbosity' : 0,\n    'random_state' : 42\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:27.635845Z","iopub.execute_input":"2022-08-13T12:39:27.636510Z","iopub.status.idle":"2022-08-13T12:39:27.646369Z","shell.execute_reply.started":"2022-08-13T12:39:27.636477Z","shell.execute_reply":"2022-08-13T12:39:27.645550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def XGBmodel(X_train,X_valid,y_train,y_valid,params):\n    matrix_train = xgb.DMatrix(X_train,label=y_train)\n    matrix_valid = xgb.DMatrix(X_valid,label=y_valid)\n    model=xgb.train(params=params,\n                    dtrain=matrix_train,num_boost_round=5000, \n                    early_stopping_rounds=10,evals=[(matrix_valid,'valid')])\n    return model\n\nmodel = XGBmodel(X_train,X_valid,y_train,y_valid,params)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:39:27.647567Z","iopub.execute_input":"2022-08-13T12:39:27.648323Z","iopub.status.idle":"2022-08-13T12:48:27.880480Z","shell.execute_reply.started":"2022-08-13T12:39:27.648271Z","shell.execute_reply":"2022-08-13T12:48:27.878997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = model.predict(xgb.DMatrix(X_test), ntree_limit = model.best_ntree_limit).tolist()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:48:27.882094Z","iopub.execute_input":"2022-08-13T12:48:27.882977Z","iopub.status.idle":"2022-08-13T12:48:28.015138Z","shell.execute_reply.started":"2022-08-13T12:48:27.882938Z","shell.execute_reply":"2022-08-13T12:48:28.013976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submition","metadata":{}},{"cell_type":"markdown","source":"### Model 6: Neural Network","metadata":{}},{"cell_type":"code","source":"subm =  pd.read_csv('../input/new-york-city-taxi-fare-prediction/sample_submission.csv')\nsubm.fare_amount = y\nsubm.to_csv('submission.csv',index=False)\nprint(\"Submitted\")","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:48:28.016561Z","iopub.execute_input":"2022-08-13T12:48:28.016936Z","iopub.status.idle":"2022-08-13T12:48:28.067142Z","shell.execute_reply.started":"2022-08-13T12:48:28.016902Z","shell.execute_reply":"2022-08-13T12:48:28.065986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fscores = pd.DataFrame({'X': list(model.get_fscore().keys()), 'Y': list(model.get_fscore().values())})\nfscores.sort_values(by='Y').plot.bar(x='X')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:48:28.068434Z","iopub.execute_input":"2022-08-13T12:48:28.068848Z","iopub.status.idle":"2022-08-13T12:48:28.503680Z","shell.execute_reply.started":"2022-08-13T12:48:28.068818Z","shell.execute_reply":"2022-08-13T12:48:28.502548Z"},"trusted":true},"execution_count":null,"outputs":[]}]}