{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport os\nimport seaborn as sns\nimport math\nimport matplotlib.pyplot as plt\nfrom lightgbm import LGBMRegressor\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import r2_score\nfrom sklearn.metrics import mean_squared_error\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-08T07:59:22.657069Z","iopub.execute_input":"2022-07-08T07:59:22.657382Z","iopub.status.idle":"2022-07-08T07:59:25.795942Z","shell.execute_reply.started":"2022-07-08T07:59:22.657289Z","shell.execute_reply":"2022-07-08T07:59:25.795137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train=pd.read_csv('../input/nyc-taxi-trip-duration/train.zip')\ndf_train.info()\ndf_train","metadata":{"execution":{"iopub.status.busy":"2022-07-08T07:59:25.797704Z","iopub.execute_input":"2022-07-08T07:59:25.798034Z","iopub.status.idle":"2022-07-08T07:59:32.591104Z","shell.execute_reply.started":"2022-07-08T07:59:25.797998Z","shell.execute_reply":"2022-07-08T07:59:32.590395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **1. Data cleaning**\n- Fill The Missing Data\n- Remove The Noisy Data\n\nNYcity data:\n- Check null\n- Check informative data (Count store_and_fwd_flag (N/Y))\n- Delete row if passenger_count=0","metadata":{}},{"cell_type":"code","source":"df_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T07:59:32.592738Z","iopub.execute_input":"2022-07-08T07:59:32.593223Z","iopub.status.idle":"2022-07-08T07:59:33.122345Z","shell.execute_reply.started":"2022-07-08T07:59:32.593187Z","shell.execute_reply":"2022-07-08T07:59:33.121650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['store_and_fwd_flag'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T07:59:33.124382Z","iopub.execute_input":"2022-07-08T07:59:33.124791Z","iopub.status.idle":"2022-07-08T07:59:33.403182Z","shell.execute_reply.started":"2022-07-08T07:59:33.124756Z","shell.execute_reply":"2022-07-08T07:59:33.402471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.scatterplot(x='pickup_longitude',y='trip_duration',data=df_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T07:59:33.405114Z","iopub.execute_input":"2022-07-08T07:59:33.405701Z","iopub.status.idle":"2022-07-08T07:59:38.082157Z","shell.execute_reply.started":"2022-07-08T07:59:33.405659Z","shell.execute_reply":"2022-07-08T07:59:38.081522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2022-07-08T07:59:38.085942Z","iopub.execute_input":"2022-07-08T07:59:38.088114Z","iopub.status.idle":"2022-07-08T07:59:38.121930Z","shell.execute_reply.started":"2022-07-08T07:59:38.088068Z","shell.execute_reply":"2022-07-08T07:59:38.121291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train= df_train.loc[df_train['trip_duration']<1000000]","metadata":{"execution":{"iopub.status.busy":"2022-07-08T07:59:38.125893Z","iopub.execute_input":"2022-07-08T07:59:38.127729Z","iopub.status.idle":"2022-07-08T07:59:38.284529Z","shell.execute_reply.started":"2022-07-08T07:59:38.127693Z","shell.execute_reply":"2022-07-08T07:59:38.283782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.scatterplot(x='pickup_latitude',y='trip_duration',data=df_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T07:59:38.288701Z","iopub.execute_input":"2022-07-08T07:59:38.290685Z","iopub.status.idle":"2022-07-08T07:59:42.962141Z","shell.execute_reply.started":"2022-07-08T07:59:38.290647Z","shell.execute_reply":"2022-07-08T07:59:42.961464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.scatterplot(x='dropoff_longitude',y='trip_duration',data=df_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T07:59:42.963226Z","iopub.execute_input":"2022-07-08T07:59:42.964524Z","iopub.status.idle":"2022-07-08T07:59:47.448290Z","shell.execute_reply.started":"2022-07-08T07:59:42.964485Z","shell.execute_reply":"2022-07-08T07:59:47.447567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.scatterplot(x='dropoff_latitude',y='trip_duration',data=df_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T07:59:47.451250Z","iopub.execute_input":"2022-07-08T07:59:47.451477Z","iopub.status.idle":"2022-07-08T07:59:52.021108Z","shell.execute_reply.started":"2022-07-08T07:59:47.451453Z","shell.execute_reply":"2022-07-08T07:59:52.020293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['passenger_count'].unique()\ndf_train= df_train.loc[df_train['passenger_count']!=0]\ndf_train","metadata":{"execution":{"iopub.status.busy":"2022-07-08T07:59:52.022345Z","iopub.execute_input":"2022-07-08T07:59:52.023078Z","iopub.status.idle":"2022-07-08T07:59:52.207601Z","shell.execute_reply.started":"2022-07-08T07:59:52.023037Z","shell.execute_reply":"2022-07-08T07:59:52.206790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.Data Integration \nUsing Data Migration Tools such as Oracle Data Service Integrator and Microsoft SQL etc\n\nNYcity data:\n- Weather to extract rain, cloud, humidity\n\n","metadata":{}},{"cell_type":"markdown","source":"# 3.Data Reduction\n* Dimensionality Reduction: Reducing the number of attributes in the dataset.\n* Numerosity Reduction: Replacing the original data volume by smaller forms of data representation.\n* Data Compression: Compressed representation of the original data.\n\nNYcity data:\n- Drop column ID, \"dropoff time\" ( because duration and pickup time ) \n- Datetime: object->datetime-> hour, day of week, month\n- store_and_fwd_flag column: string to numeric\n- lat long to distance by haversine function\n","metadata":{}},{"cell_type":"code","source":"df_train.drop('id',inplace=True,axis=1)\ndf_train.drop('dropoff_datetime',inplace=True,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T07:59:52.209021Z","iopub.execute_input":"2022-07-08T07:59:52.209279Z","iopub.status.idle":"2022-07-08T07:59:52.402097Z","shell.execute_reply.started":"2022-07-08T07:59:52.209247Z","shell.execute_reply":"2022-07-08T07:59:52.401352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[\"pickup_datetime\"]=pd.to_datetime(df_train[\"pickup_datetime\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-08T07:59:52.403427Z","iopub.execute_input":"2022-07-08T07:59:52.403663Z","iopub.status.idle":"2022-07-08T07:59:52.868749Z","shell.execute_reply.started":"2022-07-08T07:59:52.403633Z","shell.execute_reply":"2022-07-08T07:59:52.868011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['day_of_week'] = df_train['pickup_datetime'].dt.day_name()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T07:59:52.869948Z","iopub.execute_input":"2022-07-08T07:59:52.870194Z","iopub.status.idle":"2022-07-08T07:59:53.381092Z","shell.execute_reply.started":"2022-07-08T07:59:52.870160Z","shell.execute_reply":"2022-07-08T07:59:53.380302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['hour_of_the_day']=df_train['pickup_datetime'].dt.hour\ndf_train['month']=df_train['pickup_datetime'].dt.month","metadata":{"execution":{"iopub.status.busy":"2022-07-08T07:59:53.382450Z","iopub.execute_input":"2022-07-08T07:59:53.382699Z","iopub.status.idle":"2022-07-08T07:59:53.651451Z","shell.execute_reply.started":"2022-07-08T07:59:53.382667Z","shell.execute_reply":"2022-07-08T07:59:53.650749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['day_of_week']=df_train['day_of_week'].map({'Monday':1,'Tuesday':2,'Wednesday':3,'Thursday':4,'Friday':5,'Saturday':6,'Sunday':7})\ndf_train['store_and_fwd_flag']=df_train['store_and_fwd_flag'].map({'N':0,'Y':1})\ndf_train","metadata":{"execution":{"iopub.status.busy":"2022-07-08T07:59:53.652715Z","iopub.execute_input":"2022-07-08T07:59:53.652968Z","iopub.status.idle":"2022-07-08T07:59:54.035939Z","shell.execute_reply.started":"2022-07-08T07:59:53.652937Z","shell.execute_reply":"2022-07-08T07:59:54.035246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def haversine(lat1, lon1, lat2, lon2):\n    # distance between latitudes and longitudes\n    dLat = (lat2 - lat1) * math.pi / 180.0\n    dLon = (lon2 - lon1) * math.pi / 180.0\n \n    # convert to radians\n    lat1 = (lat1) * math.pi / 180.0\n    lat2 = (lat2) * math.pi / 180.0\n \n    # apply formulae\n    a = (pow(math.sin(dLat / 2), 2) +\n         pow(math.sin(dLon / 2), 2) *\n             math.cos(lat1) * math.cos(lat2));\n    rad = 6371\n    c = 2 * math.asin(math.sqrt(a))\n    return rad * c","metadata":{"execution":{"iopub.status.busy":"2022-07-08T07:59:54.037084Z","iopub.execute_input":"2022-07-08T07:59:54.037732Z","iopub.status.idle":"2022-07-08T07:59:54.045647Z","shell.execute_reply.started":"2022-07-08T07:59:54.037695Z","shell.execute_reply":"2022-07-08T07:59:54.044851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['distance']=df_train.apply(lambda row:haversine(row['pickup_latitude'],row['pickup_longitude'],row['dropoff_latitude'],row['dropoff_longitude']),axis=1)\ndf_train['distance']=df_train['distance'].astype(float)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T07:59:54.046779Z","iopub.execute_input":"2022-07-08T07:59:54.047473Z","iopub.status.idle":"2022-07-08T08:00:44.303389Z","shell.execute_reply.started":"2022-07-08T07:59:54.047439Z","shell.execute_reply":"2022-07-08T08:00:44.302498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.drop('pickup_longitude',inplace=True,axis=1)\ndf_train.drop('pickup_latitude',inplace=True,axis=1)\ndf_train.drop('dropoff_longitude',inplace=True,axis=1)\ndf_train.drop('dropoff_latitude',inplace=True,axis=1)\ndf_train.drop('pickup_datetime',inplace=True,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T08:00:44.304575Z","iopub.execute_input":"2022-07-08T08:00:44.305316Z","iopub.status.idle":"2022-07-08T08:00:44.602281Z","shell.execute_reply.started":"2022-07-08T08:00:44.305277Z","shell.execute_reply":"2022-07-08T08:00:44.601355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4.Data Transformation\n- Smoothing: Removing noise from data using clustering, regression techniques, etc.\n- Aggregation: Summary operations are applied to data.\n- Normalization: Scaling of data to fall within a smaller range.\n- Discretization: Raw values of numeric data are replaced by intervals. (for ex: age)\n\nNYCity data:\n\n- Normalization: trip_duration -> log(trip_duration)\n","metadata":{}},{"cell_type":"code","source":"sns.distplot(np.log(df_train['trip_duration'].values))\ndf_train['trip_duration']=np.log(df_train['trip_duration'])","metadata":{"execution":{"iopub.status.busy":"2022-07-08T08:00:44.604125Z","iopub.execute_input":"2022-07-08T08:00:44.604839Z","iopub.status.idle":"2022-07-08T08:00:50.909991Z","shell.execute_reply.started":"2022-07-08T08:00:44.604801Z","shell.execute_reply":"2022-07-08T08:00:50.909244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2022-07-08T08:00:50.911191Z","iopub.execute_input":"2022-07-08T08:00:50.911457Z","iopub.status.idle":"2022-07-08T08:00:50.932141Z","shell.execute_reply.started":"2022-07-08T08:00:50.911425Z","shell.execute_reply":"2022-07-08T08:00:50.931308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. Data Mining\nData Mining is a process to identify interesting patterns and knowledge from a large amount of data\n\nNYCity data: LGBMRegressor","metadata":{}},{"cell_type":"code","source":"X=df_train[['vendor_id','passenger_count','store_and_fwd_flag','day_of_week','hour_of_the_day','month','distance']]\nX","metadata":{"execution":{"iopub.status.busy":"2022-07-08T08:00:50.933780Z","iopub.execute_input":"2022-07-08T08:00:50.934063Z","iopub.status.idle":"2022-07-08T08:00:50.982325Z","shell.execute_reply.started":"2022-07-08T08:00:50.934032Z","shell.execute_reply":"2022-07-08T08:00:50.981648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y=df_train[['trip_duration']]\ny","metadata":{"execution":{"iopub.status.busy":"2022-07-08T08:00:50.983980Z","iopub.execute_input":"2022-07-08T08:00:50.984248Z","iopub.status.idle":"2022-07-08T08:00:50.997875Z","shell.execute_reply.started":"2022-07-08T08:00:50.984214Z","shell.execute_reply":"2022-07-08T08:00:50.997071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, x_test, y_train, y_test = train_test_split(X, y, test_size=0.1)\n# x_train","metadata":{"execution":{"iopub.status.busy":"2022-07-08T08:00:50.999277Z","iopub.execute_input":"2022-07-08T08:00:50.999721Z","iopub.status.idle":"2022-07-08T08:00:51.185821Z","shell.execute_reply.started":"2022-07-08T08:00:50.999678Z","shell.execute_reply":"2022-07-08T08:00:51.185050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = LGBMRegressor(n_estimators=500)  #n_estimators (int, optional (default=100)) – Number of boosted trees to fit.\nm.fit(x_train,y_train)\nprint(m.score(x_train,y_train)) #m score in this model is r2 score","metadata":{"execution":{"iopub.status.busy":"2022-07-08T08:00:51.187270Z","iopub.execute_input":"2022-07-08T08:00:51.187566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 6.Pattern Evaluation\nMeasures\n\nNYCity data: r2_score/ MSE\n--> This model should be compared to Google Distance Matrix API","metadata":{}},{"cell_type":"code","source":"y_pred = m.predict(x_test)\n# print(m.score(x_test,y_test))\nr2=r2_score(y_test, y_pred)\nmse = mean_squared_error(y_test, y_pred)\nprint(\"r2 score: %.2f\" % r2)\nprint(\"MSE: %.2f\" % mse)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 7. Knowledge Representation\nData visualization and knowledge representation tools are used to represent the mined data","metadata":{}},{"cell_type":"code","source":"y_test\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_ax = range(len(y_test))\nplt.figure(figsize=(20, 12))\nplt.plot(x_ax, y_test, label=\"original\")\nplt.plot(x_ax, y_pred, label=\"predicted\")\nplt.title(\"NY dataset test and predicted data\")\nplt.xlabel('X')\nplt.ylabel('time')\nplt.grid(True)\nplt.show() ","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}