{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport os\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns # visualization library\nimport matplotlib.pyplot as plt # visualization library\n\nimport warnings\nwarnings.filterwarnings('ignore')\n%matplotlib inline\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":false},"cell_type":"markdown","source":"## Data Loading"},{"metadata":{"trusted":true,"_uuid":"67f2550b1181c50c9d2d51208f711d5928adf2e6"},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv')\ntest = pd.read_csv('../input/test.csv')\nsample = pd.read_csv('../input/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"96feac88d8db1917997cf4e8c8a1e27cb3f6f6ef"},"cell_type":"code","source":"train.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9d390f976f53e1cbb62d6a9af7a2dc7a65276d8b"},"cell_type":"code","source":"train.tail(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aaf2fa0c7ee11e57874978ba75284f2dbad184ad"},"cell_type":"code","source":"train.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"497365474f55503e584f2595a7a0910d1d200242"},"cell_type":"code","source":"for x in train.keys():\n    print(x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4c388fddb9ef0a29881df7a73ff5ea19bad67ede"},"cell_type":"code","source":"train.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"acfc31ce899fca9c142089066b71859bc54600cd"},"cell_type":"markdown","source":"## Data Exploration"},{"metadata":{"_uuid":"7bc42b2e8a5828c9d8c93ed45c0313eb606f70b5"},"cell_type":"markdown","source":"In a business point of view, we can firstly said :\n\n    - the driver who extends the trip to earn more money (however, there is no taxi_fare column so no)\n    - the geographical position of the people between the beginning and the end of the race\n    - the time at which people took the taxi (e.g. during traffic jams or not; during the day or not etc.)\n\n"},{"metadata":{"trusted":true,"_uuid":"da53cca3c070707e3a470430f1a4546f40fb7048"},"cell_type":"code","source":"from datetime import datetime\n\ntrain['pickup_datetime'] = train['pickup_datetime'].astype('datetime64[ns]')\ntrain['dropoff_datetime'] = train['dropoff_datetime'].astype('datetime64[ns]')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a2f3265f2017acba6012fc47822a2cc41f98bf65"},"cell_type":"code","source":"pick_features = ['pickup_datetime', 'dropoff_datetime', 'vendor_id']\npick_df = train[pick_features].copy(True)\npick_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a0de1dce7001d6249c8d5379322f13b7eabfec3b"},"cell_type":"code","source":"# Pull out the month, the week,day of week and hour of day and make a new feature for each\n\npick_df['week'] = pick_df.loc[:,'pickup_datetime'].dt.week;\npick_df['weekday'] = pick_df.loc[:,'pickup_datetime'].dt.weekday;\npick_df['hour'] = pick_df.loc[:,'pickup_datetime'].dt.hour;\npick_df['month'] = pick_df.loc[:,'pickup_datetime'].dt.month;\n\n# Count number of pickups made per month and hour of day\nmonth_usage = pd.value_counts(pick_df['month']).sort_index()\nhour_usage = pd.value_counts(pick_df['hour']).sort_index()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7c91ab33f69f686d5da41e019a41dac6736d0575"},"cell_type":"code","source":"figure = plt.subplot(2, 1, 2)\nhour_usage.plot.bar(alpha = 0.5, color = 'orange')\nplt.title('Pickups over Hour of Day', fontsize = 20)\nplt.xlabel('hour', fontsize = 18)\nplt.ylabel('Count', fontsize = 18)\nplt.xticks(rotation=0)\nplt.yticks(fontsize = 18)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fd882250455c34588f96faf3322188f11b971748"},"cell_type":"code","source":"figure = plt.subplot(2, 1, 2)\nmonth_usage.plot.bar(alpha = 0.5, color = 'pink')\nplt.title('Pickups over Month', fontsize = 20)\nplt.xlabel('Month', fontsize = 18)\nplt.ylabel('Count', fontsize = 18)\nplt.xticks(rotation=0)\nplt.yticks(fontsize = 18)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"24f47fdf490d52f15ddf91b575411705b52ea4b3"},"cell_type":"markdown","source":"The pick hours of taxi trip are between 5 PM to 8 PM. During the night (from 12 AM to 7 AM) there is less taxi trip In terms of months, there are approximately as many users from January to June\n"},{"metadata":{"trusted":true,"_uuid":"a6dd7be430d151034128347b624853541a1d90d4"},"cell_type":"code","source":"train.passenger_count.min()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0397636897e61b131efa65dcf915a21d5c568975"},"cell_type":"code","source":"train.passenger_count.max()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cca5ef130c6ebd354a395ee4ce89e1d75b82d003"},"cell_type":"markdown","source":"There is 0 to 9 passengers by taxi trip. We will later drop the taxi trip with 0 passengers (because there must be atleast 1 passenger)\n"},{"metadata":{"trusted":true,"_uuid":"f53f5f212908628eb37da02e1dc10408187d27cd"},"cell_type":"code","source":"train.plot.scatter(x='pickup_longitude',y='pickup_latitude')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a3c2bff9a71c533f09aae261af6ac586a2456418"},"cell_type":"code","source":"train.plot.scatter(x='dropoff_longitude',y='dropoff_latitude')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bc782e097310068d39966421b57b4a53b317d052"},"cell_type":"code","source":"train.trip_duration.min()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b5d9d32f7218c2fc307a75593bce6f2aabd7dcd5"},"cell_type":"code","source":"train.trip_duration.max()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3d13c2ce7f561b8886af00b3f971d52538eff210"},"cell_type":"markdown","source":"The trip duration's range is between 1 sec to 3526282 sec We will later adjust this range\n"},{"metadata":{"_uuid":"acf5ce849df31112dc29b23786273b041e64aaaf"},"cell_type":"markdown","source":"## Data preprocessing"},{"metadata":{"_uuid":"60156987693b76ef5b7cc50e2433504038bb6d31"},"cell_type":"markdown","source":"### Outliers"},{"metadata":{"trusted":true,"_uuid":"d40b87cb0c7457439311aa89dd5a1d089c6492d1"},"cell_type":"code","source":"train.boxplot(figsize=(15,10))\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ed52db626e3aa7c090e20b257b732cf8914e20f8"},"cell_type":"code","source":"# As said before, there is no need to have the min (0 passenger), we will drop it\ntrain = train[train['passenger_count']>= 1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3c859f6562581c78b1a05d393eab52eb7d6351f9"},"cell_type":"code","source":"# The trip duration's range is between 1 sec to 3526282 sec\n# We will drop values that are inferior to 1 min (60 sec) and superior to 166 min (10 000 sec).\ntrain = train[train['trip_duration']>= 1 ]\ntrain = train[train['trip_duration']<= 10000 ]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"94bc218eefba40b30994011ff832b52f94f10546"},"cell_type":"code","source":"# We will drop the longitude and latitude (in pickup and dropoff that looks like outliers)\ntrain = train.loc[train['pickup_longitude']> -90]\ntrain = train.loc[train['pickup_latitude']< 47.5]\n\ntrain = train.loc[train['dropoff_longitude']> -90]\ntrain = train.loc[train['dropoff_latitude']> 34]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e5bcd86b12fc914d6e9a84f5e8d856752d4d2469"},"cell_type":"markdown","source":"## Features engineering"},{"metadata":{"trusted":true,"_uuid":"5a4823a627b8bb04999ecd2d53e512a390f6f67e"},"cell_type":"code","source":"col_diff = list(set(train.columns).difference(set(test.columns)))\ncol_diff","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"37f3a4d58036f1d6c62bfc20a8b7b19e7d2a6855"},"cell_type":"code","source":"# To use the pickup and dropoff location, we will calculate the distance between them\ntrain['dist'] = abs((train['pickup_latitude']-train['dropoff_latitude'])\n                        + (train['pickup_longitude']-train['dropoff_longitude']))\ntest['dist'] = abs((test['pickup_latitude']-test['dropoff_latitude'])\n                        + (test['pickup_longitude']-test['dropoff_longitude']))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2c35a02c55d5b94a521fd6eb8f2c948a74018f3a"},"cell_type":"code","source":"y = train[\"trip_duration\"]  # This is our target\nX = train[[\"passenger_count\",\"vendor_id\", \"pickup_longitude\", \"pickup_latitude\", \"dropoff_longitude\",\"dropoff_latitude\", \"dist\" ]]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"53b92dd275ad5ef6d24728ac3bd9bc0d69b07649"},"cell_type":"markdown","source":"## Model Selection"},{"metadata":{"trusted":true,"_uuid":"54e74f1a4636ea466e313f0e7834cfdfef2c5cd2"},"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nfrom sklearn.model_selection import ShuffleSplit\nfrom sklearn.model_selection import cross_val_score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"17bdd8f442fbf172b6da69303101a7ccf71d854c"},"cell_type":"code","source":"randf = RandomForestRegressor()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a53ff04a2f68284a88d566045118ce4d0b896e7a"},"cell_type":"code","source":"randf.fit(X, y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b52a3108c68171a4f06c960f0393e3ed109baa52"},"cell_type":"code","source":"shuffle = ShuffleSplit(n_splits=5, train_size=0.5, test_size=0.25, random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7449081df12c28a32d202ee1e856625eebc1f407"},"cell_type":"code","source":"cv_score = cross_val_score(randf, X, y, cv=shuffle, scoring='neg_mean_squared_log_error')\nfor i in range(len(cv_score)):\n    cv_score[i] = np.sqrt(abs(cv_score[i]))\nprint(np.mean(cv_score))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"aba3ee9cd20fdc4fb15a26a478feba52c70b7d69"},"cell_type":"markdown","source":"## Prediction"},{"metadata":{"trusted":true,"_uuid":"2df1e3d90e4f6f36280383d800370725e485194a"},"cell_type":"code","source":"test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bf5d89387d2d276d1013deeec67b96f42c5d44b8"},"cell_type":"code","source":"X_test = test[[\"vendor_id\", \"passenger_count\",\"pickup_longitude\", \"pickup_latitude\",\"dropoff_longitude\",\"dropoff_latitude\",\"dist\"]]\nprediction = randf.predict(X_test)\nprediction","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"194e2fd49ae03b53770ceffa67461a94931fad43"},"cell_type":"code","source":"my_submission = pd.DataFrame({'id': test.id, 'trip_duration': prediction})\nmy_submission.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bebcf597ad41d3f9bd35760696548b6d99e9c3b0"},"cell_type":"code","source":"my_submission.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}