{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn import metrics\nimport xgboost as xgb\n\ndef rmsle(y_true, y_prediction, isExponentialConversion = True):\n    if isExponentialConversion:\n        y_true = np.exp(y_true)\n        y_prediction = np.exp(y_prediction)\n        \n    log_true = np.nan_to_num(np.log(y_true + 1))\n    log_prediction = np.nan_to_num(np.log(y_prediction + 1))\n    \n    return np.sqrt(np.mean((log_true - log_prediction) ** 2))\n\n\npath = '/kaggle/input/bike-sharing-demand/'\n\ntrain = pd.read_csv(path + 'train.csv')\ntest = pd.read_csv(path + 'test.csv')\nsubmission = pd.read_csv(path + 'sampleSubmission.csv')\n\n# feature engineering\ntrain = train[train['weather'] != 4]\n\nall_data = pd.concat([train, test], ignore_index = True)\n\nall_data['datetime'] = pd.to_datetime(all_data['datetime']) #type casting\nall_data['year'] = all_data['datetime'].dt.year\nall_data['month'] = all_data['datetime'].dt.month\nall_data['hour'] = all_data['datetime'].dt.hour\nall_data['weekday'] = all_data['datetime'].dt.weekday\nall_data = all_data.drop(['datetime', 'casual', 'registered', 'windspeed', 'month'], axis = 1)\n\n\nX_train = all_data[~pd.isnull(all_data['count'])]\nX_test = all_data[pd.isnull(all_data['count'])]\n\nX_train = X_train.drop(['count'], axis = 1)\nX_test = X_test.drop(['count'], axis = 1)\n\ny = train['count']\n\n# model\nmodel = xgb.XGBRegressor()\nparameters = {'random_state': [42], 'n_estimators': [350, 370, 390], 'max_depth': [4], 'learning_rate': [0.1, 0.15]}\nrmsle_scorer = metrics.make_scorer(rmsle, greater_is_better = False)\ngridsearch_model = GridSearchCV(estimator = model, param_grid = parameters, scoring = rmsle_scorer, cv = 5)\n\nlog_y = np.log(y)\ngridsearch_model.fit(X_train, log_y)\n\nprint('최적 하이퍼파라미터: ', gridsearch_model.best_params_)\n\npredictions = gridsearch_model.best_estimator_.predict(X_test)\n# print(f'RMSLE: {rmsle(log_y, predictions):.4f}')\n\nsubmission['count'] = np.exp(predictions)\nsubmission.to_csv('submission.csv', index = False)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-27T09:47:50.827151Z","iopub.execute_input":"2022-07-27T09:47:50.827858Z","iopub.status.idle":"2022-07-27T09:48:31.165530Z","shell.execute_reply.started":"2022-07-27T09:47:50.827764Z","shell.execute_reply":"2022-07-27T09:48:31.164250Z"},"trusted":true},"execution_count":null,"outputs":[]}]}