{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\ndata_path = '/kaggle/input/bike-sharing-demand/'\n\ntrain = pd.read_csv(data_path + 'train.csv')\ntest = pd.read_csv(data_path + 'test.csv')\nsubmission = pd.read_csv(data_path + 'sampleSubmission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:05:41.803430Z","iopub.execute_input":"2022-07-23T13:05:41.804275Z","iopub.status.idle":"2022-07-23T13:05:41.902391Z","shell.execute_reply.started":"2022-07-23T13:05:41.804170Z","shell.execute_reply":"2022-07-23T13:05:41.901418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. 베이스라인 모델\n## 4-1. 피처 엔지니어링","metadata":{}},{"cell_type":"code","source":"# 이상치 제거_weather에서 4가 아닌 데이터만 추출\ntrain = train[train['weather'] != 4]\n\n#데이터 합치기\nall_data = pd.concat([train, test], ignore_index=True)\nall_data","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:24:54.461287Z","iopub.execute_input":"2022-07-23T13:24:54.462329Z","iopub.status.idle":"2022-07-23T13:24:54.493698Z","shell.execute_reply.started":"2022-07-23T13:24:54.462292Z","shell.execute_reply":"2022-07-23T13:24:54.492827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 파생 피처 추가\nfrom datetime import datetime\n\n# 날짜 피처 생성\nall_data['date'] = all_data['datetime'].apply(lambda x: x.split()[0])\n# 연도 피처 생성\nall_data['year'] = all_data['datetime'].apply(lambda x: x.split()[0].split('-')[0])\n# 월 피처 생성\nall_data['month'] = all_data['datetime'].apply(lambda x: x.split()[0].split('-')[1])\n# 시 피처 생성\nall_data['hour'] = all_data['datetime'].apply(lambda x: x.split()[1].split(':')[0])\n# 요일 피처 생성\nall_data['weekday'] = all_data['date'].apply(lambda dateString: datetime.strptime(dateString, \"%Y-%m-%d\").weekday())\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:24:56.816218Z","iopub.execute_input":"2022-07-23T13:24:56.816590Z","iopub.status.idle":"2022-07-23T13:24:57.000210Z","shell.execute_reply.started":"2022-07-23T13:24:56.816559Z","shell.execute_reply":"2022-07-23T13:24:56.999001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#필요없는 피처 제거\ndrop_features = ['casual', 'registered','datetime', 'date', 'windspeed', 'month']\n\nall_data = all_data.drop(drop_features, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:25:04.825816Z","iopub.execute_input":"2022-07-23T13:25:04.826182Z","iopub.status.idle":"2022-07-23T13:25:04.838539Z","shell.execute_reply.started":"2022-07-23T13:25:04.826153Z","shell.execute_reply":"2022-07-23T13:25:04.837161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:25:06.420491Z","iopub.execute_input":"2022-07-23T13:25:06.420884Z","iopub.status.idle":"2022-07-23T13:25:06.436322Z","shell.execute_reply.started":"2022-07-23T13:25:06.420849Z","shell.execute_reply":"2022-07-23T13:25:06.435014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#데이터를 train 데이터와 test 데이터로 나누기\n\nX_train = all_data[~pd.isnull(all_data['count'])]\nX_test = all_data[pd.isnull(all_data['count'])]\n\n#타깃값 count 제거\nX_train = X_train.drop(['count'], axis=1)\nX_test = X_test.drop(['count'], axis=1)\n\ny=train['count']","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:25:10.961487Z","iopub.execute_input":"2022-07-23T13:25:10.962638Z","iopub.status.idle":"2022-07-23T13:25:10.974799Z","shell.execute_reply.started":"2022-07-23T13:25:10.962600Z","shell.execute_reply":"2022-07-23T13:25:10.973792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:25:12.182054Z","iopub.execute_input":"2022-07-23T13:25:12.182710Z","iopub.status.idle":"2022-07-23T13:25:12.195109Z","shell.execute_reply.started":"2022-07-23T13:25:12.182674Z","shell.execute_reply":"2022-07-23T13:25:12.194366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4-2. 평가지표 계산 함수 작성","metadata":{}},{"cell_type":"code","source":"import numpy as np\n\ndef rmsle(y_true, y_pred, convertExp=True): #y_true는 실제 타깃값, y_pred는 예측값\n    #지수변환\n    if convertExp:\n        y_true = np.exp(y_true)\n        y_pred = np.exp(y_pred)\n    \n    #로그변환 후 결측값을 0으로 변환 : np.nan_to_num()함수는 결측값을 모두 0으로 바꾸는 기능.\n    log_true = np.nan_to_num(np.log(y_true+1))\n    log_pred = np.nan_to_num(np.log(y_pred+1))\n    \n    #RMSLE 계산\n    output = np.sqrt(np.mean((log_true - log_pred)**2))\n    return output","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:30:21.350422Z","iopub.execute_input":"2022-07-23T13:30:21.350787Z","iopub.status.idle":"2022-07-23T13:30:21.357758Z","shell.execute_reply.started":"2022-07-23T13:30:21.350758Z","shell.execute_reply":"2022-07-23T13:30:21.356534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4-3. 모델훈련","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\n\nlinear_reg_model = LinearRegression()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:31:23.196062Z","iopub.execute_input":"2022-07-23T13:31:23.196431Z","iopub.status.idle":"2022-07-23T13:31:23.374333Z","shell.execute_reply.started":"2022-07-23T13:31:23.196403Z","shell.execute_reply":"2022-07-23T13:31:23.373504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_y=np.log(y) #타깃값 로그변환\nlinear_reg_model.fit(X_train, log_y) #모델 훈련","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:32:00.949997Z","iopub.execute_input":"2022-07-23T13:32:00.950822Z","iopub.status.idle":"2022-07-23T13:32:00.987771Z","shell.execute_reply.started":"2022-07-23T13:32:00.950781Z","shell.execute_reply":"2022-07-23T13:32:00.986487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4-4. 모델 성능 검증","metadata":{}},{"cell_type":"code","source":"preds = linear_reg_model.predict(X_train) #예측을 수행하는 코드\nprint (f'선형회귀의 RMSLE 값 : {rmsle(log_y, preds, True):.4f}')","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:38:46.285096Z","iopub.execute_input":"2022-07-23T13:38:46.285474Z","iopub.status.idle":"2022-07-23T13:38:46.313251Z","shell.execute_reply.started":"2022-07-23T13:38:46.285444Z","shell.execute_reply":"2022-07-23T13:38:46.312000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4-5. 예측 및 결과 제출","metadata":{}},{"cell_type":"code","source":"#테스트 데이터로 예측 결과 이용해야함\n#예측한 값에 지수변환 해줘야함.\n\nlinearreg_preds = linear_reg_model.predict(X_test)\n\nsubmission['count'] = np.exp(linearreg_preds)\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:41:05.628278Z","iopub.execute_input":"2022-07-23T13:41:05.629306Z","iopub.status.idle":"2022-07-23T13:41:05.682561Z","shell.execute_reply.started":"2022-07-23T13:41:05.629267Z","shell.execute_reply":"2022-07-23T13:41:05.681188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. 성능개선1. 릿지 회귀 모델","metadata":{}},{"cell_type":"markdown","source":"## 5-1. 하이퍼파라미터 최적화(모델훈련)\n* 앞선 데이스라인 모델의 '데이터 불러오기'-'피처 엔지니어링'-'평가지표 계산 함수 작성'까지 같음.\n* 모델 훈련 단계에서 그리드서치 기법 사용\n* 그리드서치 : 하이퍼파라미터를 격자처럼 촘촘하게 순회하며 최적의 하이퍼파라미터값을 찾는 기법","metadata":{}},{"cell_type":"code","source":"#기본 릿지 모델 생성\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn import metrics\n\nridge_model = Ridge()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:52:33.507031Z","iopub.execute_input":"2022-07-23T13:52:33.507418Z","iopub.status.idle":"2022-07-23T13:52:33.513333Z","shell.execute_reply.started":"2022-07-23T13:52:33.507383Z","shell.execute_reply":"2022-07-23T13:52:33.512072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#하이퍼파라미터 값 목록\nridge_params = {'max_iter':[3000], 'alpha':[0.1, 1, 2, 3, 4, 10, 30, 100, 200, 300, 400, 800, 900, 1000]}\n\n#교차검증용 평가함수(RMSLE 점수 계산)\nrmsle_scorer = metrics.make_scorer(rmsle, greater_is_better=False)\n\n#그리드서치 객체 생성\ngridsearch_ridge_model = GridSearchCV(estimator=ridge_model,#릿지 모델\n                                     param_grid=ridge_params, #값 목록을 딕셔너리로 \n                                     scoring=rmsle_scorer, #평가 지표\n                                     cv=5) #교차 검증 분할 수\n\n#그리드서치 수행\nlog_y=np.log(y) #타깃값 로그 변환\ngridsearch_ridge_model.fit(X_train, log_y) #그리드서치 훈련\n\nprint('최적 하이퍼파라미터:', gridsearch_ridge_model.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:58:09.606323Z","iopub.execute_input":"2022-07-23T13:58:09.606729Z","iopub.status.idle":"2022-07-23T13:58:11.675832Z","shell.execute_reply.started":"2022-07-23T13:58:09.606694Z","shell.execute_reply":"2022-07-23T13:58:11.674535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5-2. 성능검증","metadata":{}},{"cell_type":"code","source":"#예측\npreds = gridsearch_ridge_model.best_estimator_.predict(X_train)\n#평가\nprint (f'릿지 회귀 RMSLE 값: {rmsle(log_y, preds, True):.4f}')","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:00:07.721377Z","iopub.execute_input":"2022-07-23T14:00:07.721771Z","iopub.status.idle":"2022-07-23T14:00:07.752354Z","shell.execute_reply.started":"2022-07-23T14:00:07.721742Z","shell.execute_reply":"2022-07-23T14:00:07.750885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 6. 성능개선2: 라쏘 회귀 모델\n* 앞선 데이스라인 모델의 '데이터 불러오기'-'피처 엔지니어링'-'평가지표 계산 함수 작성'까지 같음.","metadata":{}},{"cell_type":"markdown","source":"## 6-1. 하이퍼파라미터 최적화(모델훈련)","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import Lasso\n\n#모델 생성\nlasso_model = Lasso()\n\n#하이퍼파라미터 값 목록\nlasso_alpha = 1/np.array([0.1, 1, 2, 3, 4, 10, 30, 100, 200, 300, 400, 800, 900, 1000])\nlasso_params = {'max_iter':[3000], 'alpha':lasso_alpha}\n\n#그리드서치 객체 생성\ngridsearch_lasso_model = GridSearchCV(estimator=lasso_model,\n                                     param_grid=lasso_params,\n                                     scoring=rmsle_scorer,\n                                     cv=5)\n\n#그리드서치 수행\nlog_y=np.log(y)\ngridsearch_lasso_model.fit(X_train, log_y)\n\nprint('최적 하이퍼파라미터:', gridsearch_lasso_model.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:05:26.617863Z","iopub.execute_input":"2022-07-23T14:05:26.618247Z","iopub.status.idle":"2022-07-23T14:05:32.340467Z","shell.execute_reply.started":"2022-07-23T14:05:26.618218Z","shell.execute_reply":"2022-07-23T14:05:32.339101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 6-2. 성능검증","metadata":{}},{"cell_type":"code","source":"#예측\npreds=gridsearch_lasso_model.best_estimator_.predict(X_train)\n\n#평가\nprint (f'라쏘 회귀 RMSLE 값: {rmsle(log_y, preds, True):.4f}')","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:07:07.947525Z","iopub.execute_input":"2022-07-23T14:07:07.947902Z","iopub.status.idle":"2022-07-23T14:07:07.974783Z","shell.execute_reply.started":"2022-07-23T14:07:07.947872Z","shell.execute_reply":"2022-07-23T14:07:07.973118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 7. 성능개선3: 랜덤 포레스트 회귀 모델\n* 훈련 데이터를 랜덤하게 샘플링한 모델 n개를 각각 훈련하여 결과를 평균하는 방법\n* 앞선 데이스라인 모델의 '데이터 불러오기'-'피처 엔지니어링'-'평가지표 계산 함수 작성'까지 같음.","metadata":{}},{"cell_type":"markdown","source":"## 7-1. 하이퍼파라미터 최적화(모델훈련)","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\n\n#모델 생성\nrandomforest_model=RandomForestRegressor()\n\n#그리드서치 객체 생성\nrf_params = {'random_state':[42], 'n_estimators':[100, 120, 140]}\ngridsearch_random_forest_model = GridSearchCV(estimator=randomforest_model,\n                                             param_grid = rf_params,\n                                             scoring=rmsle_scorer,\n                                             cv=5)\n\n#그리드서치 수행\nlog_y=np.log(y)\ngridsearch_random_forest_model.fit(X_train, log_y)\nprint('최적의 하이퍼파라미터: ', gridsearch_random_forest_model.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:11:30.389270Z","iopub.execute_input":"2022-07-23T14:11:30.389655Z","iopub.status.idle":"2022-07-23T14:12:19.177378Z","shell.execute_reply.started":"2022-07-23T14:11:30.389625Z","shell.execute_reply":"2022-07-23T14:12:19.176075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 7-2. 모델 성능 검증","metadata":{}},{"cell_type":"code","source":"#예측\npreds=gridsearch_random_forest_model.best_estimator_.predict(X_train)\n\n#평가\nprint (f'랜덤포레스트 회귀 RMSLE 값: {rmsle(log_y, preds, True):.4f}')","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:12:50.542077Z","iopub.execute_input":"2022-07-23T14:12:50.543112Z","iopub.status.idle":"2022-07-23T14:12:50.856964Z","shell.execute_reply.started":"2022-07-23T14:12:50.543075Z","shell.execute_reply":"2022-07-23T14:12:50.855620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"randomforest_preds=gridsearch_random_forest_model.best_estimator_.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:15:17.349805Z","iopub.execute_input":"2022-07-23T14:15:17.350240Z","iopub.status.idle":"2022-07-23T14:15:17.551525Z","shell.execute_reply.started":"2022-07-23T14:15:17.350207Z","shell.execute_reply":"2022-07-23T14:15:17.550604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#랜덤포레스트 결과가 제일 좋으므로 제출\nsubmission['count'] = np.exp(randomforest_preds) #지수변환\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:15:47.604761Z","iopub.execute_input":"2022-07-23T14:15:47.605165Z","iopub.status.idle":"2022-07-23T14:15:47.631645Z","shell.execute_reply.started":"2022-07-23T14:15:47.605133Z","shell.execute_reply":"2022-07-23T14:15:47.630700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}