{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n    \ndata_path = '/kaggle/input/bike-sharing-demand/'    \n\ntrain = pd.read_csv(data_path +'train.csv')\ntest = pd.read_csv(data_path + 'test.csv')\nsubmission = pd.read_csv(data_path + 'sampleSubmission.csv')\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-28T15:36:23.624336Z","iopub.execute_input":"2022-07-28T15:36:23.624769Z","iopub.status.idle":"2022-07-28T15:36:23.678529Z","shell.execute_reply.started":"2022-07-28T15:36:23.624737Z","shell.execute_reply":"2022-07-28T15:36:23.677075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 성능 개선1; 릿지 회귀 모델","metadata":{}},{"cell_type":"markdown","source":"릿지; 훈련 데이터에 모델이 과대적합 되지 않도록 규제해줌\n\n1) 데이터 불러오기\n\n2) 피쳐 엔지니어링\n\n3) 평가지표 계산 함수 작성 (기본 베이스라인 모델을 만들 때 사용했던 평가지표 계산 함수 재사용)\n\n4) 하이퍼파라미터 최적화 (모델 훈련; 릿지 회기로 모델 생성, 그리드서치* 객체 생성, 그리드서치로 훈련)\n\n5) 성능 검증\n\n6) 제출\n\n**그리드서치; 하이퍼파라미터를 격자처럼 촘촘히 순회하며 최적의 하이퍼파라미터 값을 찾는 기법. 각 하이퍼파라미터를 적용한 모델마다 교차검증하여 최종적으로 가장 좋은 성능의 하이퍼팔미터 값을 찾음. 테스트하려는 하이퍼파라미터 값의 범위만 전달하면 알아서 모든 가능한 조합을 순회하며 교차 검증*","metadata":{}},{"cell_type":"code","source":"# 이상치 제거 (폭우 폭설날 대여 건)\ntrain = train[train['weather'] != 4]\n\n# 데이터 합치기\nall_data = pd.concat([train, test], ignore_index=True)\nall_data","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:36:23.680541Z","iopub.execute_input":"2022-07-28T15:36:23.681843Z","iopub.status.idle":"2022-07-28T15:36:23.725432Z","shell.execute_reply.started":"2022-07-28T15:36:23.681765Z","shell.execute_reply":"2022-07-28T15:36:23.723811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 릿지 파생 피쳐 추가","metadata":{}},{"cell_type":"code","source":"from datetime import datetime\n\nall_data['date'] = all_data['datetime'].apply(lambda x : x.split()[0]) #찐 날짜를 하려면 년월일이 나와야 함\nall_data['year'] = all_data['datetime'].apply(lambda x : x.split()[0].split('-')[0]) #연도별\nall_data['month'] = all_data['datetime'].apply(lambda x : x.split()[0].split('-')[1]) #월별\nall_data['hour'] = all_data['datetime'].apply(lambda x : x.split()[1].split(':')[0]) #시간대별\nall_data[\"weekday\"] = all_data['date'].apply(lambda dateString : datetime.strptime(dateString, \"%Y-%m-%d\").weekday()) # 요일","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:36:23.731738Z","iopub.execute_input":"2022-07-28T15:36:23.732169Z","iopub.status.idle":"2022-07-28T15:36:24.027317Z","shell.execute_reply.started":"2022-07-28T15:36:23.732129Z","shell.execute_reply":"2022-07-28T15:36:24.026313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drop_features = ['casual', 'registered', 'datetime', 'date', 'windspeed', 'month'] # 불필요한 피쳐 제거\nall_data = all_data.drop(drop_features, axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:36:24.029219Z","iopub.execute_input":"2022-07-28T15:36:24.029585Z","iopub.status.idle":"2022-07-28T15:36:24.043599Z","shell.execute_reply.started":"2022-07-28T15:36:24.029553Z","shell.execute_reply":"2022-07-28T15:36:24.041997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = all_data[~pd.isnull(all_data['count'])]\nX_test = all_data[pd.isnull(all_data['count'])]\n\nX_train = X_train.drop(['count'], axis = 1)\nX_test = X_test.drop(['count'], axis = 1)\n\ny = train['count']","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:36:24.045662Z","iopub.execute_input":"2022-07-28T15:36:24.046100Z","iopub.status.idle":"2022-07-28T15:36:24.062329Z","shell.execute_reply.started":"2022-07-28T15:36:24.046036Z","shell.execute_reply":"2022-07-28T15:36:24.060842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:36:24.064786Z","iopub.execute_input":"2022-07-28T15:36:24.065747Z","iopub.status.idle":"2022-07-28T15:36:24.086110Z","shell.execute_reply.started":"2022-07-28T15:36:24.065705Z","shell.execute_reply":"2022-07-28T15:36:24.085175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 릿지 평가지표 계산 함수 작성 ( rmsle라는 함수(클래스) 생성)","metadata":{}},{"cell_type":"code","source":"import numpy as np\ndef rmsle(y_true, y_pred, convertExp = True): #y_true는 실제 타깃값, y_pred는 예측값\n    # 지수변환 함수 = convertExp, exp()\n    # log(count)를 타깃값으로 사용하기 때문에 미리 지수변환 \n    # 즉, rmsle에 던져지는 값들이 앞으로의 계산에 바로 사용될 수 있는 상태가 아니라 정규분포 만들려고 로그취해진 값임\n    if convertExp:\n        y_true = np.exp(y_true)\n        y_pred = np.exp(y_pred)\n\n    # RMSLE 계산용 로그변환(0일 때 -무한대로 갈 수 있으니까 +1인 채로 로그변환) 후 결측값을 0으로 변환(np.nan_to_num)\n    log_true = np.nan_to_num(np.log(y_true+1))\n    log_pred = np.nan_to_num(np.log(y_pred+1))\n    \n    # RMSLE 계산\n    output = np.sqrt(np.mean((log_true - log_pred)**2)) # 로그변환은 그냥 하는 것이 아니라 +1해서 한다고 위에서 정의\n    return output","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:36:24.087555Z","iopub.execute_input":"2022-07-28T15:36:24.088167Z","iopub.status.idle":"2022-07-28T15:36:24.096026Z","shell.execute_reply.started":"2022-07-28T15:36:24.088120Z","shell.execute_reply":"2022-07-28T15:36:24.094774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 릿지 모델 생성","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import Ridge\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn import metrics\n\nridge_model = Ridge()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:36:24.097842Z","iopub.execute_input":"2022-07-28T15:36:24.098351Z","iopub.status.idle":"2022-07-28T15:36:24.108178Z","shell.execute_reply.started":"2022-07-28T15:36:24.098312Z","shell.execute_reply":"2022-07-28T15:36:24.106918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 릿지 그리드서치 객체 생성\n\n하이퍼파라미터 값을 바꾸어가며 모델 성능 교차검증으로 평가하기 위해서는 \n\n1) 비교 검증해볼 하이퍼파라밑 값 목록\n\n2) 대상모델 -> 이미 제작 완\n\n3) 교차검증용 평가 수단 (평가 함수)\n\n를 알아야 함. ","metadata":{}},{"cell_type":"code","source":"# 하이퍼 파라미터 값 목록: 릿지 모델에서는 alpha가 중요한데, 값이 클수록 규제 강도 세짐\nridge_params = {'max_iter': [3000], 'alpha': [0.1, 1, 2, 3, 4, 10, 30, 100, 200, 300, 400, 800, 900, 1000]}\n\n# 교차검증용 평가함수 (RMSLE 점수 계산)\nrmsle_scorer = metrics.make_scorer(rmsle, greater_is_better = False)\n\n# 그리드서치(with릿지) 객체 생성\ngridsearch_ridge_model = GridSearchCV(estimator = ridge_model,  # 릿지 모델 (분류 및 회귀 모델)\n                                     param_grid = ridge_params, # 값 목록 (딕셔너리 형태로 하이퍼파리미터명과 여러 하이퍼파라미터 값을 지정, 위에서 max_iter나 alpha 값 넣은 것)\n                                     scoring = rmsle_scorer,    # 평가지표 (사이킷런에서 기본적인 평가지표를 문자열 형태로 제공하지만, 여기서는 RSMLE 지표 만들어서 넣음)\n                                     cv = 5)                    # 교차검증 분할 수","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:36:24.110615Z","iopub.execute_input":"2022-07-28T15:36:24.111358Z","iopub.status.idle":"2022-07-28T15:36:24.122053Z","shell.execute_reply.started":"2022-07-28T15:36:24.111318Z","shell.execute_reply":"2022-07-28T15:36:24.121100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 릿지 그리드서치 수행","metadata":{}},{"cell_type":"code","source":"log_y = np.log(y)\ngridsearch_ridge_model.fit(X_train, log_y)\nprint('최적 하이퍼파라미터:', gridsearch_ridge_model.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:36:24.127906Z","iopub.execute_input":"2022-07-28T15:36:24.128630Z","iopub.status.idle":"2022-07-28T15:36:26.929378Z","shell.execute_reply.started":"2022-07-28T15:36:24.128590Z","shell.execute_reply":"2022-07-28T15:36:26.928009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 릿지 성능 검증","metadata":{}},{"cell_type":"code","source":"#예측\npreds = gridsearch_ridge_model.best_estimator_.predict(X_train)\n\n#평가\nprint(f'릿지 회귀 RMSLE 값 : {rmsle(log_y, preds, True):.4f}')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:36:26.932111Z","iopub.execute_input":"2022-07-28T15:36:26.936963Z","iopub.status.idle":"2022-07-28T15:36:26.982234Z","shell.execute_reply.started":"2022-07-28T15:36:26.936886Z","shell.execute_reply":"2022-07-28T15:36:26.980776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 성능 개선2; 라쏘 회귀 모델","metadata":{}},{"cell_type":"markdown","source":"# 라쏘 하이퍼파라미터 최적화 (모델 훈련)","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import Lasso\n\nlasso_model = Lasso() # 모델 생성\n\nlasso_alpha = 1/np.array([0.1, 1, 2, 3, 4, 10, 30, 100, 200, 300, 400, 800, 900, 1000])\nlasso_params ={'max_iter': [3000], 'alpha': lasso_alpha}\n\n# 그리드서치(with 라소) 객체생성\ngridsearch_lasso_model = GridSearchCV(estimator = lasso_model, \n                                      param_grid = lasso_params, \n                                      scoring = rmsle_scorer, \n                                      cv = 5)\n\nlog_y = np.log(y)\ngridsearch_lasso_model.fit(X_train, log_y)\n\nprint('최적 하이퍼파라미터: ', gridsearch_lasso_model.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:36:26.984574Z","iopub.execute_input":"2022-07-28T15:36:26.985539Z","iopub.status.idle":"2022-07-28T15:36:33.243889Z","shell.execute_reply.started":"2022-07-28T15:36:26.985478Z","shell.execute_reply":"2022-07-28T15:36:33.242466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 라쏘 성능 검증","metadata":{}},{"cell_type":"code","source":"preds = gridsearch_lasso_model.best_estimator_.predict(X_train)\n\nprint(f'라쏘 회귀 RMSLE 값: {rmsle(log_y, preds, True): .4f}')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:36:33.251946Z","iopub.execute_input":"2022-07-28T15:36:33.252957Z","iopub.status.idle":"2022-07-28T15:36:33.313975Z","shell.execute_reply.started":"2022-07-28T15:36:33.252895Z","shell.execute_reply":"2022-07-28T15:36:33.311445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 성능 개선3; 랜덤포레스트 모델","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor # 모델 생성\n\nrandomforest_model = RandomForestRegressor()\n\nrf_params = {'random_state': [42], 'n_estimators':[100, 120, 140]}\n# random_state: 랜덤 시드값, n_estimators: 랜덤포레스트를 구성하는 결정 트리 개수\ngridsearch_random_forest_model = GridSearchCV(estimator = randomforest_model, \n                                              param_grid = rf_params, \n                                              scoring = rmsle_scorer, \n                                              cv = 5)\n\nlog_y = np.log(y)\ngridsearch_random_forest_model.fit(X_train, log_y)\nprint('최적 하이퍼파라미터: ', gridsearch_random_forest_model.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:36:33.319358Z","iopub.execute_input":"2022-07-28T15:36:33.323055Z","iopub.status.idle":"2022-07-28T15:37:23.606975Z","shell.execute_reply.started":"2022-07-28T15:36:33.322963Z","shell.execute_reply":"2022-07-28T15:37:23.605802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 랜덤포레스트 모델 성능 검증","metadata":{}},{"cell_type":"code","source":"preds = gridsearch_random_forest_model.best_estimator_.predict(X_train)\n\nprint(f'랜덤포레스트 회귀 RMSLE 값: {rmsle(log_y, preds, True): .4f}')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:37:23.608560Z","iopub.execute_input":"2022-07-28T15:37:23.608925Z","iopub.status.idle":"2022-07-28T15:37:23.937996Z","shell.execute_reply.started":"2022-07-28T15:37:23.608892Z","shell.execute_reply":"2022-07-28T15:37:23.936602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 세 모델 중 성능이 가장 좋은 모델의 테스트파일 예측 결과 제출","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nrandomforest_preds = gridsearch_random_forest_model.best_estimator_.predict(X_test)\n\nfigure, axes = plt.subplots(ncols = 2)\nfigure.set_size_inches(10, 4)\n\nsns.histplot(y, bins = 50, ax = axes[0])\naxes[0].set_title('Train Data Distribution')\nsns.histplot(np.exp(randomforest_preds), bins = 50, ax = axes[1]) # 다시 지수함수로 로그 풀어주기\naxes[1].set_title('Predicted Test Data Distribution')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:44:21.771433Z","iopub.execute_input":"2022-07-28T15:44:21.774057Z","iopub.status.idle":"2022-07-28T15:44:22.584305Z","shell.execute_reply.started":"2022-07-28T15:44:21.773925Z","shell.execute_reply":"2022-07-28T15:44:22.582340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['count'] = np.exp(randomforest_preds)\nsubmission.to_csv('submission.csv', index = False) ","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:47:37.260613Z","iopub.execute_input":"2022-07-28T15:47:37.261025Z","iopub.status.idle":"2022-07-28T15:47:37.307428Z","shell.execute_reply.started":"2022-07-28T15:47:37.260994Z","shell.execute_reply":"2022-07-28T15:47:37.306375Z"},"trusted":true},"execution_count":null,"outputs":[]}]}