{"cells":[{"metadata":{"_uuid":"cd33a8b14aa1500a3465a01dd34a1d0a2fd89276"},"cell_type":"markdown","source":"**This kernel is for applicated Machine Learning to predicting trip duration in NYC taxi**\n**Please to check my kernel [NYC EDA](https://www.kaggle.com/baoanh/nyc-taxi-duration-eda-by-nguyen-khac-bao-anh) to understanding this dataset and knowing kind of dataset that i used for this kernel**"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# visualiser les données\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n# Any results you write to the current directory are saved as output.\n\nimport warnings\nwarnings.filterwarnings('ignore')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"52b64e9c52d61311452db16e9b944eb933dd42e6"},"cell_type":"code","source":"%matplotlib inline\nsns.set({'figure.figsize':(16,8)})","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":false},"cell_type":"markdown","source":"# Data Loading"},{"metadata":{"trusted":true,"_uuid":"7958f86e205f7a1c7ef5905d8ae95bcd37349ae8"},"cell_type":"code","source":"train = pd.read_csv(\"../input/nyc-taxi-duration-eda-by-nguyen-khac-bao-anh/training_data.csv\")\ntest = pd.read_csv(\"../input/nyc-taxi-duration-eda-by-nguyen-khac-bao-anh/testing_data.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d156370f100b3041de7bce7cd3d4e1a997e485bb"},"cell_type":"code","source":"print(f\"shape of training set{train.shape}\")\nprint(f\"shape of testing set{test.shape}\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"928bd1aff8d791d46678e024b49744786e7322de"},"cell_type":"markdown","source":"### Data Training vs Data testing:"},{"metadata":{"trusted":true,"_uuid":"fcc431efb1edb4974b6b23f9448f53346707c030"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d4c5569e98ea61ce4f93afd171dd484955c54f75"},"cell_type":"code","source":"test.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c37a5a0ad53c222b6e1c760d3eb7b84e66f75941"},"cell_type":"markdown","source":"# Select dataset"},{"metadata":{"trusted":true,"_uuid":"f37178d3fec0d23db4affbc642a1afbc1c38115c"},"cell_type":"code","source":"col_diff = list(set(train.columns).difference(set(test.columns)))\nprint(f\"La différence de la variable entre data training et data testing:\\\n{set(train.columns).difference(set(test.columns))}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e62e08c543c633d8530994fe6bbeb48d4875634a"},"cell_type":"code","source":"xtrain = train.drop(['id', 'pickup_datetime', 'dropoff_datetime', 'trip_duration', 'log_trip_duration'], axis = 1).as_matrix()\nxtest = test.drop(['id', 'pickup_datetime', ], axis = 1).as_matrix()\ny = train['log_trip_duration'].values\ndel(train, test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"35b2a5d861ad03deaf5699aad438577fd9335e4f"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split, cross_val_score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6f155314b096c2d283ee776bfdcd16b5f96185ad"},"cell_type":"code","source":"X_train, X_valid, y_train, y_valid = train_test_split(xtrain,y, test_size=0.2, random_state=42)\nX_train.shape, X_valid.shape, y_train.shape, y_valid.shape","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"29ed3c9ad099a565ff1d072bfc5564c08b448bec"},"cell_type":"markdown","source":"# Select model"},{"metadata":{"trusted":true,"_uuid":"322fb2a0cbec03d2afe39fdf9f7c657229008430"},"cell_type":"code","source":"#from sklearn.ensemble import RandomForestRegressor","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6ba13c98b8e2e3ce2c0d25abf10db4e5e7cefd4d"},"cell_type":"code","source":"#rf_defaut = RandomForestRegressor()\n# crosse validation pour tester si le model est stable \n#rf_cv = cross_val_score(rf_defaut, X_train, y_train, cv=5)\n#rf_cv","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5c7d8eab7fb97035f0e51888eaea191ddd85f62f"},"cell_type":"code","source":"#plt.plot(range(1,len(rf_cv)+1), rf_cv)\n#plt.ylim(np.min(rf_cv)-0.1,np.max(rf_cv)+0.1)\n#plt.xlabel(\"nombre de fold\")\n#plt.ylabel(\"score du model Random Forest Regressor\");","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"204e55b07935c8e61ae54676148f23eccb4f07c5"},"cell_type":"markdown","source":"> le line chart au-dessus, il nous dit que ce model est stable, c'est à dire, ce model est capable de généraliser"},{"metadata":{"trusted":true,"_uuid":"584f165e3258e18dea1741430e82f9e3746fc0f3"},"cell_type":"code","source":"# la fonction permet de nous donner un score qui est le règle de cette compétition\n# (root mean squared log error)\n#from sklearn.metrics import mean_squared_log_error, mean_squared_error\ndef rmse(y,pred):\n    return np.sqrt(np.mean(np.square(np.log(np.exp(y))-np.log(np.exp(pred)))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e08601ee77fea3a1335b9549b827c1ea53e52c74"},"cell_type":"code","source":"#rf_defaut = RandomForestRegressor()\n#rf_defaut.fit(X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"de70a6271918acfec47b41cb81200a5222a6f22d"},"cell_type":"code","source":"#y_pred = rf_defaut.predict(X_valid)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fe6f3643e7b15e40aa47a5ef36dbb63d1a5f2087"},"cell_type":"code","source":"#print(rmse(y_valid,y_pred))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0e0d805f0630e1468cb8bc8f18a6b17f26ac8c01"},"cell_type":"markdown","source":"## Grid search pour optimiser l'algorithme Random Forest"},{"metadata":{"trusted":true,"_uuid":"603ccfae423707e14f876f1727d94465b9d1a982"},"cell_type":"code","source":"#from sklearn.model_selection import GridSearchCV","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"226240db8d9aee12e5c7ccdbf5b289d97d999509"},"cell_type":"code","source":"# n_estimators et max_depth pour fitter bien le model mais causer overfitting\n# par contre, min_samples_leaf et min_samples_split nous permets de éviter overfitting en donnant la valeur\n# plus grand\n#params = {\n#    'n_estimators' : [10, 15, 20],\n#    'max_depth': [30, 50, 100],\n#    'min_samples_leaf': [100],\n#    'min_samples_split': [150]\n#}\n#rf2 = RandomForestRegressor()\n#gs_rf2 = GridSearchCV(rf2, param_grid=params, scoring='neg_mean_squared_error',cv=3, verbose=10, n_jobs=-1)\n#gs_rf2.fit(X_train, y_train)\n#gs_rf2.best_score_\n#best_rf2 = gs_rf2.best_estimator_","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"25fb6991b8395ba63af1818f906a8a6b33755f18"},"cell_type":"code","source":"#rf2 = RandomForestRegressor(n_estimators=10,min_samples_leaf=100, min_samples_split=150)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a6b2c174f90171d880d02562d1e3028c27f84d6f"},"cell_type":"code","source":"#rf2_cv = cross_val_score(rf2, X_train, y_train, cv=5)\n#rf2_cv","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8a124574ce580ad175d7622af25212883b8d1e1a"},"cell_type":"code","source":"#plt.plot(range(1,len(rf2_cv)+1), rf2_cv)\n#plt.ylim(np.min(rf2_cv)-0.1,np.max(rf2_cv)+0.1)\n#plt.xlabel(\"nombre de fold\")\n#plt.ylabel(\"score du model Random Forest Regressor avec hyperparameters\");","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"830077ac8925b02b01e2b03e633c20592c79a4d4"},"cell_type":"code","source":"#rf2.fit(X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a83f7f410ab5862c0ddf469c58aeb7c31cc375c3"},"cell_type":"code","source":"#pred = rf2.predict(X_valid)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"57a3074cd567d7d4a9932087bc5a68d5c84af0ba"},"cell_type":"code","source":"#print(rmse(y_valid,pred))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"52ce170c27aff2956b8b8abf5e570fc11d8392a0"},"cell_type":"markdown","source":"# LightGBM pour prédire trip duration"},{"metadata":{"trusted":true,"_uuid":"68feeb26caa1b52aa543f9f317dc9466a869d596"},"cell_type":"code","source":"import lightgbm as lgb","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"809c9f3e6babdf468b22eca4a97c1d31aefd77c4"},"cell_type":"code","source":"#lgb_train = lgb.Dataset(X_train, y_train)\n#lgb_valid = lgb.Dataset(X_valid, y_valid)\n# training all dataset\ndtrain = lgb.Dataset(xtrain,y)\ndel(X_train, y_train, X_valid, y_valid, xtrain,y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c8729e557b114898cda0a9ce516ac50ff75228fd"},"cell_type":"code","source":"lgb_params = {\n    'learning_rate': 0.1,\n    'max_depth': 8,\n    'num_leaves': 55, \n    'objective': 'regression',\n    #'metric': {'rmse'},\n    'feature_fraction': 0.9,\n    'bagging_fraction': 0.5,\n    #'bagging_freq': 5,\n    'max_bin': 300}       # 1000","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"344a7957b52a918112ae875de0cd0369664c0695"},"cell_type":"code","source":"#cv_result_lgb = lgb.cv(lgb_params,\n#                       lgb_train, \n#                       num_boost_round=1000, \n#                       nfold=3,\n#                       early_stopping_rounds=50, \n#                       verbose_eval=100, \n#                       show_stdv=True,stratified=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e2e8dd2855b57006da088caabc493ee18e838a1f"},"cell_type":"code","source":"#n_rounds = len(cv_result_lgb['rmsle-mean'])\n#print('num_boost_rounds_lgb=' + str(n_rounds))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2e2832c2ab38924325fa93b2e05a0366489889a9"},"cell_type":"code","source":"# visualisation des résultat dans cv\n# CV scores\n#train_scores = np.array(cv_result_lgb['rmsle-mean'])\n#train_stds = np.array(cv_result_lgb['rmsle-stdv'])\n#plt.plot(train_scores, color='violet')\n#plt.fill_between(range(len(cv_result_lgb['rmsle-mean'])), \n#                 train_scores - train_stds, train_scores + train_stds, \n#                 alpha=0.1, color='violet')\n#plt.title('LightGMB CV-results')\n#plt.xlabel(\"number of rounds\")\n#plt.ylabel(\"score\");","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"05f323245a06a76c4720bb4425855740e578b8bf"},"cell_type":"code","source":"# Train a model\n#model_lgb = lgb.train(lgb_params, \n#                      dtrain, \n#                      feval=lgb_rmsle_score, \n#                      num_boost_round=n_rounds)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"698d9a4ba4fb63c956d80b906ba5afb87e45f133"},"cell_type":"code","source":"## Predict on train\n#y_train_pred = model_lgb.predict(X_train)\n#print('RMSLE on train = {}'.format(rmse(y_train_pred, y_train)))\n## Predict on validation\n#y_valid_pred = model_lgb.predict(X_valid)\n#print('RMSLE on valid = {}'.format(rmse(y_valid_pred, y_valid)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4acedf606e59a608047c2984f620364156cbf023"},"cell_type":"code","source":"# Train a model\nmodel_lgb = lgb.train(lgb_params, \n                      dtrain,\n                      num_boost_round=1500)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"85a1a4c01d13c079ce250f7e45df385c45999724"},"cell_type":"markdown","source":"# submission on kaggle"},{"metadata":{"trusted":true,"_uuid":"c01b30ad3b4f3bd1a3c03b36e8d6830fdc84e4a1"},"cell_type":"code","source":"submit = pd.read_csv('../input/nyc-taxi-trip-duration/sample_submission.csv')\nsubmit.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0cefff62d9a66063117bbca8c9ec1e97d1eaa053"},"cell_type":"code","source":"pred_test = np.exp(model_lgb.predict(xtest))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5b2c3c6291d9fb51fa6eddb99407109e5e012a68"},"cell_type":"code","source":"submit['trip_duration'] = pred_test","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"29ce21df3b89121dc3dabd76ff669aebca70b101"},"cell_type":"code","source":"submit.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b0b3da1286537e5e69f9c3497489bb5be2e3dd02"},"cell_type":"code","source":"submit.to_csv(\"submit_file.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ffd5a59ee339d30e1fb7d064caf4218a3a6713e5"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}