{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport matplotlib\nfrom sklearn.model_selection import train_test_split\nimport xgboost as xgb ","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:38:16.595149Z","iopub.execute_input":"2022-08-05T18:38:16.596160Z","iopub.status.idle":"2022-08-05T18:38:16.601478Z","shell.execute_reply.started":"2022-08-05T18:38:16.596110Z","shell.execute_reply":"2022-08-05T18:38:16.600503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/new-york-city-taxi-fare-prediction/train.csv', nrows = 5000000)\ninfo = train.info()\n\ndf_description = train.describe()\n\nprint(train.columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:38:16.613367Z","iopub.execute_input":"2022-08-05T18:38:16.613690Z","iopub.status.idle":"2022-08-05T18:38:28.466906Z","shell.execute_reply.started":"2022-08-05T18:38:16.613661Z","shell.execute_reply":"2022-08-05T18:38:28.465759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.set_index(\"key\")\n\ntrain[['date', 'time', 'timezone']] = train['pickup_datetime'].str.split(' ', expand=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:38:28.469307Z","iopub.execute_input":"2022-08-05T18:38:28.469754Z","iopub.status.idle":"2022-08-05T18:38:47.151940Z","shell.execute_reply.started":"2022-08-05T18:38:28.469712Z","shell.execute_reply":"2022-08-05T18:38:47.150587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop([\"pickup_datetime\", \"timezone\"], axis=1, inplace=True)\ntrain.set_index(\"key\", inplace=True)\ntrain[['hour', 'minute', 'second']] = train['time'].str.split(':', expand=True)\ntrain[[\"hour\", \"minute\", \"second\"]] = train[[\"hour\", \"minute\", \"second\"]].astype(\"int\")\ntrain['hour'] = np.where(train['minute'] > 30 , train['hour'] + 1, train['hour'])","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:38:47.153722Z","iopub.execute_input":"2022-08-05T18:38:47.154545Z","iopub.status.idle":"2022-08-05T18:39:10.232207Z","shell.execute_reply.started":"2022-08-05T18:38:47.154510Z","shell.execute_reply":"2022-08-05T18:39:10.230990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.hour = pd.cut(train.hour,bins=[0,2,6,9,12,16,17,18,21,23, 24], labels=['Midnight', 'Early morning', 'Morning', 'Late morning', 'Afternoon', 'Late afternoon', 'Early evening', \"Evening\", \"Night\", 'Midnight'], ordered=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:39:10.235320Z","iopub.execute_input":"2022-08-05T18:39:10.236094Z","iopub.status.idle":"2022-08-05T18:39:10.436082Z","shell.execute_reply.started":"2022-08-05T18:39:10.236048Z","shell.execute_reply":"2022-08-05T18:39:10.434916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop([\"time\", \"minute\", \"second\"], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:39:10.437358Z","iopub.execute_input":"2022-08-05T18:39:10.437677Z","iopub.status.idle":"2022-08-05T18:39:10.841467Z","shell.execute_reply.started":"2022-08-05T18:39:10.437649Z","shell.execute_reply":"2022-08-05T18:39:10.840387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[['year', 'month', 'day']] = train['date'].str.split('-', expand=True)\ntrain[['year', 'month', 'day']] = train[['year', 'month', 'day']].astype(\"int\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:39:10.843868Z","iopub.execute_input":"2022-08-05T18:39:10.844710Z","iopub.status.idle":"2022-08-05T18:39:30.249180Z","shell.execute_reply.started":"2022-08-05T18:39:10.844672Z","shell.execute_reply":"2022-08-05T18:39:30.248222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.month = pd.cut(train.month,bins=[0,1,2,3,4,5,6,7,8,9,10,11,12], labels=['January', 'February', 'March', 'April', 'May', 'June', 'July', \"August\", \"September\", 'October', 'November', 'December'])","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:39:30.250423Z","iopub.execute_input":"2022-08-05T18:39:30.251275Z","iopub.status.idle":"2022-08-05T18:39:30.463090Z","shell.execute_reply.started":"2022-08-05T18:39:30.251244Z","shell.execute_reply":"2022-08-05T18:39:30.462156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.sort_values(by=\"year\")","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:39:30.464310Z","iopub.execute_input":"2022-08-05T18:39:30.464634Z","iopub.status.idle":"2022-08-05T18:39:32.231806Z","shell.execute_reply.started":"2022-08-05T18:39:30.464605Z","shell.execute_reply":"2022-08-05T18:39:32.230689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lat_distance(lat_lon1: list, lat_lon2: list):\n    p = np.pi/180\n    lat1, lon1 = lat_lon1\n    lat2, lon2 = lat_lon2\n    a = 0.5 - np.cos((lat2 - lat1) * p)/2 + np.cos(lat1 * p) * np.cos(lat2 * p) * (1 - np.cos((lon2 - lon1) * p)) / 2\n    return 12742 * np.arcsin(np.sqrt(a))\n\ntrain['distance'] = lat_distance([train.pickup_latitude, train.pickup_longitude], [train.dropoff_latitude, train.dropoff_longitude])","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:39:32.234216Z","iopub.execute_input":"2022-08-05T18:39:32.234960Z","iopub.status.idle":"2022-08-05T18:39:32.674346Z","shell.execute_reply.started":"2022-08-05T18:39:32.234917Z","shell.execute_reply":"2022-08-05T18:39:32.673209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.dropna(how='any', axis=0)\ntrain = train[train.fare_amount >= 0]\ntrain = train[train.passenger_count < 8]","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:39:32.678988Z","iopub.execute_input":"2022-08-05T18:39:32.679464Z","iopub.status.idle":"2022-08-05T18:39:36.687155Z","shell.execute_reply.started":"2022-08-05T18:39:32.679432Z","shell.execute_reply":"2022-08-05T18:39:36.685987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def label_function(val):\n    return f'{val / 100 * len(train):.0f}\\n{val:.0f}%'\n\nfig, (ax1, ax2, ax3, ax4, ax5) = plt.subplots(ncols=5, figsize=(40, 25))\n\n\ntrain.groupby('passenger_count').size().plot(kind='pie',  textprops={'fontsize': 14}, ax=ax1, autopct=label_function)\ntrain.groupby('hour').size().plot(kind='pie',  textprops={'fontsize': 14}, ax=ax2, autopct=label_function)\ntrain.groupby('year').size().plot(kind='pie',  textprops={'fontsize': 14}, ax=ax3, autopct=label_function)\ntrain.groupby('month').size().plot(kind='pie',  textprops={'fontsize': 14}, ax=ax4, autopct=label_function)\ntrain.groupby('day').size().plot(kind='pie',  textprops={'fontsize': 14}, ax=ax5, autopct=label_function)\n\n\nax1.set_xlabel('passenger_count', size=14)\nax2.set_xlabel('hour', size=14)\nax3.set_xlabel('year', size=14)\nax4.set_xlabel('month', size=14)\nax5.set_xlabel('day', size=14)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:39:36.688513Z","iopub.execute_input":"2022-08-05T18:39:36.688846Z","iopub.status.idle":"2022-08-05T18:39:38.340067Z","shell.execute_reply.started":"2022-08-05T18:39:36.688816Z","shell.execute_reply":"2022-08-05T18:39:38.339098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax2, ax3) = plt.subplots(ncols=2, figsize=(40, 25))\n\ncurrent_palette = matplotlib.colors.hex2color(\"#0089cf\")\nsns.boxplot(y=\"passenger_count\", data=train, color=current_palette, ax=ax2, orient=\"v\")\ncurrent_palette = matplotlib.colors.hex2color('#ff7f0e')\nsns.violinplot(y=\"passenger_count\", data=train, color=current_palette, ax=ax2, orient=\"v\")\n\n\ncurrent_palette = matplotlib.colors.hex2color(\"#0089cf\")\nsns.boxplot(y=\"distance\", data=train, color=current_palette, ax=ax3, orient=\"v\")\ncurrent_palette = matplotlib.colors.hex2color('#ff7f0e')\nsns.violinplot(y=\"distance\", data=train, color=current_palette, ax=ax3, orient=\"v\")","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:39:38.341345Z","iopub.execute_input":"2022-08-05T18:39:38.341838Z","iopub.status.idle":"2022-08-05T18:39:59.944786Z","shell.execute_reply.started":"2022-08-05T18:39:38.341807Z","shell.execute_reply":"2022-08-05T18:39:59.944031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nclean_corr = pd.get_dummies(train[[\"passenger_count\", \"day\", \"distance\", \"fare_amount\"]])\n\ncorr_df = clean_corr.corr()\nfiltered_df = corr_df #  [((corr_df >= .1) | (corr_df <= -.1)) & (corr_df !=1.000)]\nplt.figure(figsize=(30,10))\nsns.heatmap(filtered_df, annot=True, cmap=\"coolwarm\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:39:59.946170Z","iopub.execute_input":"2022-08-05T18:39:59.946498Z","iopub.status.idle":"2022-08-05T18:40:01.377934Z","shell.execute_reply.started":"2022-08-05T18:39:59.946468Z","shell.execute_reply":"2022-08-05T18:40:01.376585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = train[[\"fare_amount\", \"passenger_count\", \"hour\", \"month\", \"distance\"]]\n\nX_train, X_test, y_train, y_test = train_test_split(df.drop(\"fare_amount\",  axis=1), df['fare_amount'], test_size=0.3, random_state = 0)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:40:01.379416Z","iopub.execute_input":"2022-08-05T18:40:01.379775Z","iopub.status.idle":"2022-08-05T18:40:06.854248Z","shell.execute_reply.started":"2022-08-05T18:40:01.379743Z","shell.execute_reply":"2022-08-05T18:40:06.853360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"parameters = {\n    'colsample_bytree': 0.9, \n    'silent': 0,\n    'max_depth': 10,\n    'gamma' :0,\n    'objective':'reg:linear',\n    'eval_metric':'rmse',\n    'eta':.03, \n    'subsample': 1\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:40:06.855586Z","iopub.execute_input":"2022-08-05T18:40:06.856438Z","iopub.status.idle":"2022-08-05T18:40:06.861391Z","shell.execute_reply.started":"2022-08-05T18:40:06.856407Z","shell.execute_reply":"2022-08-05T18:40:06.860294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def XGBmodel(X_train,X_test,y_train,y_test,params):\n    matrix_train = xgb.DMatrix(X_train,label=y_train, enable_categorical=True)\n    matrix_test = xgb.DMatrix(X_test,label=y_test, enable_categorical=True)\n    model=xgb.train(params=parameters,\n                    dtrain=matrix_train,num_boost_round=5000, \n                    early_stopping_rounds=20,evals=[(matrix_test,'test')])\n    return model\n\nmodel = XGBmodel(X_train,X_test,y_train,y_test,parameters)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:40:06.862467Z","iopub.execute_input":"2022-08-05T18:40:06.863349Z","iopub.status.idle":"2022-08-05T18:49:58.320750Z","shell.execute_reply.started":"2022-08-05T18:40:06.863319Z","shell.execute_reply":"2022-08-05T18:49:58.319515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv(\"../input/new-york-city-taxi-fare-prediction/test.csv\", index_col=\"key\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:49:58.323158Z","iopub.execute_input":"2022-08-05T18:49:58.324446Z","iopub.status.idle":"2022-08-05T18:49:58.370058Z","shell.execute_reply.started":"2022-08-05T18:49:58.324366Z","shell.execute_reply":"2022-08-05T18:49:58.369070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[['date', 'time', 'timezone']] = test['pickup_datetime'].str.split(' ', expand=True)\ntest.drop([\"pickup_datetime\", \"timezone\"], axis=1, inplace=True)\n\ntest[['hour', 'minute', 'second']] = test['time'].str.split(':', expand=True)\ntest[[\"hour\", \"minute\", \"second\"]] = test[[\"hour\", \"minute\", \"second\"]].astype(\"int\")\ntest['hour'] = np.where(test['minute'] > 30 , test['hour'] + 1, test['hour'])\ntest.hour = pd.cut(test.hour,bins=[0,2,6,9,12,16,17,18,21,23, 24], labels=['Midnight', 'Early morning', 'Morning', 'Late morning', 'Afternoon', 'Late afternoon', 'Early evening', \"Evening\", \"Night\", 'Midnight'], ordered=False)\ntest.drop([\"time\", \"minute\", \"second\"], axis=1, inplace=True)\n\ntest[['year', 'month', 'day']] = test['date'].str.split('-', expand=True)\ntest[['year', 'month', 'day']] = test[['year', 'month', 'day']].astype(\"int\")\ntest.month = pd.cut(test.month,bins=[0,1,2,3,4,5,6,7,8,9,10,11,12], labels=['January', 'February', 'March', 'April', 'May', 'June', 'July', \"August\", \"September\", 'October', 'November', 'December'])\n\ntest['distance'] = lat_distance([test.pickup_latitude, test.pickup_longitude], [test.dropoff_latitude, test.dropoff_longitude])\n\ntest = test[[ \"passenger_count\", \"hour\", \"month\", \"distance\"]]","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:49:58.371255Z","iopub.execute_input":"2022-08-05T18:49:58.371557Z","iopub.status.idle":"2022-08-05T18:49:58.483342Z","shell.execute_reply.started":"2022-08-05T18:49:58.371530Z","shell.execute_reply":"2022-08-05T18:49:58.482437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction = model.predict(xgb.DMatrix(test, enable_categorical=True), ntree_limit = model.best_ntree_limit).tolist()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:50:52.814230Z","iopub.execute_input":"2022-08-05T18:50:52.815303Z","iopub.status.idle":"2022-08-05T18:50:52.891336Z","shell.execute_reply.started":"2022-08-05T18:50:52.815264Z","shell.execute_reply":"2022-08-05T18:50:52.890260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_pred = pd.DataFrame({'key': test.index, 'fare_amount': prediction})\nfinal_pred.to_csv('../working/submission.csv', index=False)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T19:08:19.017935Z","iopub.execute_input":"2022-08-05T19:08:19.018713Z","iopub.status.idle":"2022-08-05T19:08:19.053919Z","shell.execute_reply.started":"2022-08-05T19:08:19.018676Z","shell.execute_reply":"2022-08-05T19:08:19.052784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_pred.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:49:58.607917Z","iopub.execute_input":"2022-08-05T18:49:58.608368Z","iopub.status.idle":"2022-08-05T18:49:58.615422Z","shell.execute_reply.started":"2022-08-05T18:49:58.608335Z","shell.execute_reply":"2022-08-05T18:49:58.614188Z"},"trusted":true},"execution_count":null,"outputs":[]}]}