{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## 1. 탐색적 데이터 분석","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-23T12:38:07.158165Z","iopub.execute_input":"2022-07-23T12:38:07.158940Z","iopub.status.idle":"2022-07-23T12:38:07.164773Z","shell.execute_reply.started":"2022-07-23T12:38:07.158878Z","shell.execute_reply":"2022-07-23T12:38:07.163685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/bike-sharing-demand/train.csv')\ntest = pd.read_csv('../input/bike-sharing-demand/test.csv')\nsubmission = pd.read_csv('../input/bike-sharing-demand/sampleSubmission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:07.173423Z","iopub.execute_input":"2022-07-23T12:38:07.174260Z","iopub.status.idle":"2022-07-23T12:38:07.219815Z","shell.execute_reply.started":"2022-07-23T12:38:07.174210Z","shell.execute_reply":"2022-07-23T12:38:07.218623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape, test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:07.221592Z","iopub.execute_input":"2022-07-23T12:38:07.221894Z","iopub.status.idle":"2022-07-23T12:38:07.227829Z","shell.execute_reply.started":"2022-07-23T12:38:07.221868Z","shell.execute_reply":"2022-07-23T12:38:07.227103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 데이터 피처 확인하기\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:07.228891Z","iopub.execute_input":"2022-07-23T12:38:07.229362Z","iopub.status.idle":"2022-07-23T12:38:07.249457Z","shell.execute_reply.started":"2022-07-23T12:38:07.229331Z","shell.execute_reply":"2022-07-23T12:38:07.248610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head() # count는 예측해야 할 값, casual과 registered는 제거 필요 -> 모델 훈련 시 피처의 개수가 같아야함...?","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:07.251052Z","iopub.execute_input":"2022-07-23T12:38:07.251560Z","iopub.status.idle":"2022-07-23T12:38:07.271110Z","shell.execute_reply.started":"2022-07-23T12:38:07.251515Z","shell.execute_reply":"2022-07-23T12:38:07.270332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 피처별 결측값 및 데이터 타입 확인\ntrain.info() # 결측치 없음, 데이터타입은 다양함","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:07.272489Z","iopub.execute_input":"2022-07-23T12:38:07.273217Z","iopub.status.idle":"2022-07-23T12:38:07.288150Z","shell.execute_reply.started":"2022-07-23T12:38:07.273185Z","shell.execute_reply":"2022-07-23T12:38:07.287032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.info() # 결측치 없음, 데이터타입은 다양함","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:07.289206Z","iopub.execute_input":"2022-07-23T12:38:07.289781Z","iopub.status.idle":"2022-07-23T12:38:07.302755Z","shell.execute_reply.started":"2022-07-23T12:38:07.289737Z","shell.execute_reply":"2022-07-23T12:38:07.301923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# datetime 이용하여 날짜데이터 처리하기\ntrain['datetime'] = pd.to_datetime(train['datetime'])\ntrain.info() # 'datetime' 변수의 데이터타입이 datetime으로 변경됨","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:07.304564Z","iopub.execute_input":"2022-07-23T12:38:07.305138Z","iopub.status.idle":"2022-07-23T12:38:07.329419Z","shell.execute_reply.started":"2022-07-23T12:38:07.305107Z","shell.execute_reply":"2022-07-23T12:38:07.328616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# datetime 데이터 타입을 year, month, day, hour, minute, second, weekday로 나눠서 새로운 피처로 추가하기\ntrain['year'] = train['datetime'].dt.year\ntrain['year'] # 2011-01-20 00:00:00의 형식에서 year(연도)에 해당하는 것들만 추출하여 새 변수 생성","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:07.332030Z","iopub.execute_input":"2022-07-23T12:38:07.333072Z","iopub.status.idle":"2022-07-23T12:38:07.344883Z","shell.execute_reply.started":"2022-07-23T12:38:07.333014Z","shell.execute_reply":"2022-07-23T12:38:07.343657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 나머지 변수들도 추가\ntrain['month'] = train['datetime'].dt.month\ntrain['day'] = train['datetime'].dt.day\ntrain['hour'] = train['datetime'].dt.hour\ntrain['minute'] = train['datetime'].dt.minute\ntrain['second'] = train['datetime'].dt.second\ntrain['weekday'] = train['datetime'].dt.day_name() # day_name은 ()추가 필요\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:07.349775Z","iopub.execute_input":"2022-07-23T12:38:07.350343Z","iopub.status.idle":"2022-07-23T12:38:07.389799Z","shell.execute_reply.started":"2022-07-23T12:38:07.350310Z","shell.execute_reply":"2022-07-23T12:38:07.388710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# map 함수 활용하여 season, weather 피처를 숫자에서 문자로 변경하기\ntrain['season'] = train['season'].map({1: 'Spring', \n                                       2: 'Summer', \n                                       3: 'Fall', \n                                       4: 'Winter'})\ntrain['weather'] = train['weather'].map({1: 'Clear', \n                                         2: 'Mist, Few clouds', \n                                         3: 'Light Snow, Rain, Thunderstorm', \n                                         4: 'Heavy Rain, Thunderstorm, Snow, Fog'})\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:07.411598Z","iopub.execute_input":"2022-07-23T12:38:07.411986Z","iopub.status.idle":"2022-07-23T12:38:07.437942Z","shell.execute_reply.started":"2022-07-23T12:38:07.411957Z","shell.execute_reply":"2022-07-23T12:38:07.436652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:07.440248Z","iopub.execute_input":"2022-07-23T12:38:07.441183Z","iopub.status.idle":"2022-07-23T12:38:07.450337Z","shell.execute_reply.started":"2022-07-23T12:38:07.441148Z","shell.execute_reply":"2022-07-23T12:38:07.449493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mpl.rc('font', size = 15)\nsns.displot(train['count']) # 타겟변수의 데이터가 정규분포가 아니라 왼쪽으로 치우침 -> 이럴 때 로그변환 사용!\n# 정규분포에 가까울수록 회귀모델의 성능이 좋아진다!","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:07.452188Z","iopub.execute_input":"2022-07-23T12:38:07.452875Z","iopub.status.idle":"2022-07-23T12:38:07.777559Z","shell.execute_reply.started":"2022-07-23T12:38:07.452832Z","shell.execute_reply":"2022-07-23T12:38:07.776431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 타겟변수 로그변환 후 분포 재확인 -> 이전보다 정규분포에 근사함\nsns.displot(np.log(train['count']))","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:07.779482Z","iopub.execute_input":"2022-07-23T12:38:07.779815Z","iopub.status.idle":"2022-07-23T12:38:08.099033Z","shell.execute_reply.started":"2022-07-23T12:38:07.779785Z","shell.execute_reply":"2022-07-23T12:38:08.098120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 막대 그래프 시각화 1단계(서브플롯 설정)\nmpl.rc('font', size = 14)\nmpl.rc('axes', titlesize = 15)\nfigure, axes = plt.subplots(nrows = 3, ncols = 2)\nplt.tight_layout()\nfigure.set_size_inches(10, 9)\n# 막대 그래프 시각화 2단계(각 축에 서브플롯 할당)\nsns.barplot(x = 'year', y = 'count', data = train, ax = axes[0, 0])\nsns.barplot(x = 'month', y = 'count', data = train, ax = axes[0, 1])\nsns.barplot(x = 'day', y = 'count', data = train, ax = axes[1, 0])\nsns.barplot(x = 'hour', y = 'count', data = train, ax = axes[1, 1])\nsns.barplot(x = 'minute', y = 'count', data = train, ax = axes[2, 0])\nsns.barplot(x = 'second', y = 'count', data = train, ax = axes[2, 1])\n# 막대 그래프 시각화 3-1단계(세부 설정: 제목 달기)\naxes[0, 0].set(title = 'Rental amounts by year')\naxes[0, 1].set(title = 'Rental amounts by month')\naxes[1, 0].set(title = 'Rental amounts by day')\naxes[1, 1].set(title = 'Rental amounts by hour')\naxes[2, 0].set(title = 'Rental amounts by minute')\naxes[2, 1].set(title = 'Rental amounts by second')\n# 막대 그래프 시각화 3-2단계(세부 설정: 1행에 위치한 서브플롯의 x축 라벨 90도 회전)\naxes[1, 0].tick_params(axis = 'x', labelrotation = 90)\naxes[1, 1].tick_params(axis = 'x', labelrotation = 90)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:08.100539Z","iopub.execute_input":"2022-07-23T12:38:08.101271Z","iopub.status.idle":"2022-07-23T12:38:11.356364Z","shell.execute_reply.started":"2022-07-23T12:38:08.101223Z","shell.execute_reply":"2022-07-23T12:38:11.355241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 박스플롯 시각화 1단계(서브플롯 설정)\nfigure, axes = plt.subplots(nrows = 2, ncols = 2)\nplt.tight_layout()\nfigure.set_size_inches(10, 10)\n# 박스플롯 시각화 2단계(서브플롯 할당)\nsns.boxplot(x = 'season', y = 'count', data = train, ax = axes[0,0])\nsns.boxplot(x = 'weather', y = 'count', data = train, ax = axes[0,1])\nsns.boxplot(x = 'holiday', y = 'count', data = train, ax = axes[1,0])\nsns.boxplot(x = 'workingday', y = 'count', data = train, ax = axes[1,1])\n# 박스플롯 시각화 3-1단계(세부 설정: 제목 달기)\naxes[0,0].set(title = 'Box Plot On Count Across Season')\naxes[0,1].set(title = 'Box Plot On Count Across Weather')\naxes[1,0].set(title = 'Box Plot On Count Across Holiday')\naxes[1,1].set(title = 'Box Plot On Count Across Workingday')\n# 박스플롯 시각화 3-2단계(세부 설정: [0,1] 서브플롯의 x축 라벨 회전)\naxes[0,1].tick_params(axis = 'x', labelrotation = 10)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:11.358380Z","iopub.execute_input":"2022-07-23T12:38:11.359029Z","iopub.status.idle":"2022-07-23T12:38:11.949643Z","shell.execute_reply.started":"2022-07-23T12:38:11.358990Z","shell.execute_reply":"2022-07-23T12:38:11.948166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 포인트플롯 시각화 1단계\nmpl.rc('font', size = 11)\nfigure, axes = plt.subplots(nrows = 5)\nfigure.set_size_inches(12, 18)\n# 포인트플롯 시각화 2단계\nsns.pointplot(x = 'hour', y = 'count', hue = 'workingday', data = train, ax = axes[0])\nsns.pointplot(x = 'hour', y = 'count', hue = 'holiday', data = train, ax = axes[1])\nsns.pointplot(x = 'hour', y = 'count', hue = 'weekday', data = train, ax = axes[2])\nsns.pointplot(x = 'hour', y = 'count', hue = 'season', data = train, ax = axes[3])\nsns.pointplot(x = 'hour', y = 'count', hue = 'weather', data = train, ax = axes[4])","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:11.951295Z","iopub.execute_input":"2022-07-23T12:38:11.951756Z","iopub.status.idle":"2022-07-23T12:38:24.171334Z","shell.execute_reply.started":"2022-07-23T12:38:11.951710Z","shell.execute_reply":"2022-07-23T12:38:24.170533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 회귀선을 포함한 산점도 그래프\nmpl.rc('font', size = 15)\nfigure, axes = plt.subplots(nrows = 2, ncols = 2)\nplt.tight_layout()\nfigure.set_size_inches(7,6)\n\nsns.regplot(x = 'temp', y = 'count', data = train, ax = axes[0,0],\n            scatter_kws = {'alpha' : 0.2}, line_kws = {'color' : 'blue'})\nsns.regplot(x = 'atemp', y = 'count', data = train, ax = axes[0,1],\n            scatter_kws = {'alpha' : 0.2}, line_kws = {'color' : 'blue'})\nsns.regplot(x = 'windspeed', y = 'count', data = train, ax = axes[1,0],\n            scatter_kws = {'alpha' : 0.2}, line_kws = {'color' : 'blue'})\nsns.regplot(x = 'humidity', y = 'count', data = train, ax = axes[1,1],\n            scatter_kws = {'alpha' : 0.2}, line_kws = {'color' : 'blue'})","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:24.172666Z","iopub.execute_input":"2022-07-23T12:38:24.173176Z","iopub.status.idle":"2022-07-23T12:38:27.203810Z","shell.execute_reply.started":"2022-07-23T12:38:24.173141Z","shell.execute_reply":"2022-07-23T12:38:27.202713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 상관관계 매트릭스\ntrain[['temp', 'atemp', 'humidity', 'windspeed', 'count']].corr() # 한 눈에 확인하기 곤란함 -> 히트맵 사용\n\n# 히트맵 그리기\ncorrMat = train[['temp', 'atemp', 'humidity', 'windspeed', 'count']].corr()\nfigure, axes = plt.subplots()\nfigure.set_size_inches(10,10)\nsns.heatmap(corrMat, annot = True)\naxes.set(title = 'Heatmap of Numerical Data') # count와 가장 상관계수가 낮은 windspeed는 제거","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:27.205189Z","iopub.execute_input":"2022-07-23T12:38:27.205525Z","iopub.status.idle":"2022-07-23T12:38:27.510033Z","shell.execute_reply.started":"2022-07-23T12:38:27.205490Z","shell.execute_reply":"2022-07-23T12:38:27.508882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. 베이스라인 모델 작성","metadata":{}},{"cell_type":"code","source":"# 데이터 세팅\ntrain = pd.read_csv('../input/bike-sharing-demand/train.csv')\ntest = pd.read_csv('../input/bike-sharing-demand/test.csv')\nsubmission = pd.read_csv('../input/bike-sharing-demand/sampleSubmission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:27.512769Z","iopub.execute_input":"2022-07-23T12:38:27.513099Z","iopub.status.idle":"2022-07-23T12:38:27.554924Z","shell.execute_reply.started":"2022-07-23T12:38:27.513069Z","shell.execute_reply":"2022-07-23T12:38:27.554067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# weather의 이상치 제거\ntrain = train[train['weather'] != 4]","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:27.556306Z","iopub.execute_input":"2022-07-23T12:38:27.556824Z","iopub.status.idle":"2022-07-23T12:38:27.562723Z","shell.execute_reply.started":"2022-07-23T12:38:27.556788Z","shell.execute_reply":"2022-07-23T12:38:27.561646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train 데이터와 test 데이터 합차기\nall_data = pd.concat([train, test], ignore_index = True)\nall_data","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:27.564126Z","iopub.execute_input":"2022-07-23T12:38:27.564703Z","iopub.status.idle":"2022-07-23T12:38:27.592677Z","shell.execute_reply.started":"2022-07-23T12:38:27.564672Z","shell.execute_reply":"2022-07-23T12:38:27.591978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 날짜와 관련된 파생변수 생성\nall_data['datetime'] = pd.to_datetime(all_data['datetime'])\nall_data['year'] = all_data['datetime'].dt.year\nall_data['month'] = all_data['datetime'].dt.month\nall_data['hour'] = all_data['datetime'].dt.hour\nall_data['weekday'] = all_data['datetime'].dt.weekday # day_name은 ()추가 필요\nall_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:27.593873Z","iopub.execute_input":"2022-07-23T12:38:27.594361Z","iopub.status.idle":"2022-07-23T12:38:27.627667Z","shell.execute_reply.started":"2022-07-23T12:38:27.594331Z","shell.execute_reply":"2022-07-23T12:38:27.626933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 불필요한 피처 제거하기\ndrop_features = ['casual', 'registered', 'datetime', 'windspeed', 'month']\nall_data = all_data.drop(drop_features, axis = 1) # 피처셀렉션 완료","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:27.628868Z","iopub.execute_input":"2022-07-23T12:38:27.629363Z","iopub.status.idle":"2022-07-23T12:38:27.636618Z","shell.execute_reply.started":"2022-07-23T12:38:27.629334Z","shell.execute_reply":"2022-07-23T12:38:27.635523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 데이터 다시 나누기\nX_train = all_data[~pd.isnull(all_data['count'])]\nX_test = all_data[pd.isnull(all_data['count'])]","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:27.637908Z","iopub.execute_input":"2022-07-23T12:38:27.638774Z","iopub.status.idle":"2022-07-23T12:38:27.652596Z","shell.execute_reply.started":"2022-07-23T12:38:27.638741Z","shell.execute_reply":"2022-07-23T12:38:27.651527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = X_train.drop(['count'], axis = 1)\nX_test = X_test.drop(['count'], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:27.653754Z","iopub.execute_input":"2022-07-23T12:38:27.654607Z","iopub.status.idle":"2022-07-23T12:38:27.664325Z","shell.execute_reply.started":"2022-07-23T12:38:27.654576Z","shell.execute_reply":"2022-07-23T12:38:27.663343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = train['count']","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:27.665937Z","iopub.execute_input":"2022-07-23T12:38:27.666344Z","iopub.status.idle":"2022-07-23T12:38:27.677451Z","shell.execute_reply.started":"2022-07-23T12:38:27.666299Z","shell.execute_reply":"2022-07-23T12:38:27.676432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:27.679247Z","iopub.execute_input":"2022-07-23T12:38:27.679633Z","iopub.status.idle":"2022-07-23T12:38:27.697604Z","shell.execute_reply.started":"2022-07-23T12:38:27.679590Z","shell.execute_reply":"2022-07-23T12:38:27.696624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 모델훈련\nfrom sklearn.linear_model import LinearRegression\nlinear_reg_model = LinearRegression()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:27.699172Z","iopub.execute_input":"2022-07-23T12:38:27.699702Z","iopub.status.idle":"2022-07-23T12:38:27.703539Z","shell.execute_reply.started":"2022-07-23T12:38:27.699669Z","shell.execute_reply":"2022-07-23T12:38:27.702704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_y = np.log(y)\nlinear_reg_model.fit(X_train, log_y)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:27.704938Z","iopub.execute_input":"2022-07-23T12:38:27.706236Z","iopub.status.idle":"2022-07-23T12:38:27.727323Z","shell.execute_reply.started":"2022-07-23T12:38:27.706170Z","shell.execute_reply":"2022-07-23T12:38:27.726134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = linear_reg_model.predict(X_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:27.728830Z","iopub.execute_input":"2022-07-23T12:38:27.729772Z","iopub.status.idle":"2022-07-23T12:38:27.738770Z","shell.execute_reply.started":"2022-07-23T12:38:27.729728Z","shell.execute_reply":"2022-07-23T12:38:27.737414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rmsle(y_true, y_pred, convertExp = True) :\n    if convertExp :\n        y_true = np.exp(y_true)\n        y_pred = np.exp(y_pred)\n        \n    log_true = np.nan_to_num(np.log(y_true+1))\n    log_pred = np.nan_to_num(np.log(y_pred+1))\n    \n    output = np.sqrt(np.mean((log_true - log_pred) ** 2))\n    return output","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:27.750348Z","iopub.execute_input":"2022-07-23T12:38:27.752727Z","iopub.status.idle":"2022-07-23T12:38:27.767936Z","shell.execute_reply.started":"2022-07-23T12:38:27.752661Z","shell.execute_reply":"2022-07-23T12:38:27.766311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 모델 성능 확인 1\nprint(round(rmsle(log_y, preds, True), 4))","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:27.770938Z","iopub.execute_input":"2022-07-23T12:38:27.773115Z","iopub.status.idle":"2022-07-23T12:38:27.789957Z","shell.execute_reply.started":"2022-07-23T12:38:27.773064Z","shell.execute_reply":"2022-07-23T12:38:27.788427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 모델 성능 확인 2\n#from sklearn.metrics import mean_squared_log_error\n#preds_exp = np.exp(preds)\n#msle = mean_squared_log_error(y, preds_exp)\n#rmsle = np.sqrt(msle)\n#print(round(rmsle, 4))","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:27.793075Z","iopub.execute_input":"2022-07-23T12:38:27.794996Z","iopub.status.idle":"2022-07-23T12:38:27.803347Z","shell.execute_reply.started":"2022-07-23T12:38:27.794943Z","shell.execute_reply":"2022-07-23T12:38:27.801859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 예측 및 결과 제출\nlinearreg_preds = linear_reg_model.predict(X_test)\nsubmission['count'] = np.exp(linearreg_preds)\nsubmission.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:27.806350Z","iopub.execute_input":"2022-07-23T12:38:27.808605Z","iopub.status.idle":"2022-07-23T12:38:27.871724Z","shell.execute_reply.started":"2022-07-23T12:38:27.808549Z","shell.execute_reply":"2022-07-23T12:38:27.870264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. 성능개선","metadata":{}},{"cell_type":"code","source":"# 릿지 회귀 모델\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn import metrics\n\nridge_model = Ridge()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:27.874054Z","iopub.execute_input":"2022-07-23T12:38:27.875035Z","iopub.status.idle":"2022-07-23T12:38:27.881716Z","shell.execute_reply.started":"2022-07-23T12:38:27.874978Z","shell.execute_reply":"2022-07-23T12:38:27.880522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ridge_params = {'max_iter' : [3000], 'alpha' : [0.1,1,2,3,4,10,30,100,200,300,400,800,900,1000]}\nrmsle_scorer = metrics.make_scorer(rmsle, greater_is_better = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:27.883770Z","iopub.execute_input":"2022-07-23T12:38:27.884622Z","iopub.status.idle":"2022-07-23T12:38:27.896478Z","shell.execute_reply.started":"2022-07-23T12:38:27.884581Z","shell.execute_reply":"2022-07-23T12:38:27.895159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gridsearch_ridge_model = GridSearchCV(estimator = ridge_model, param_grid = ridge_params, scoring = rmsle_scorer, cv = 5)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:27.899026Z","iopub.execute_input":"2022-07-23T12:38:27.899989Z","iopub.status.idle":"2022-07-23T12:38:27.910851Z","shell.execute_reply.started":"2022-07-23T12:38:27.899935Z","shell.execute_reply":"2022-07-23T12:38:27.909224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_y = np.log(y)\ngridsearch_ridge_model.fit(X_train, log_y)\nprint(gridsearch_ridge_model.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:27.912327Z","iopub.execute_input":"2022-07-23T12:38:27.913045Z","iopub.status.idle":"2022-07-23T12:38:28.659567Z","shell.execute_reply.started":"2022-07-23T12:38:27.912999Z","shell.execute_reply":"2022-07-23T12:38:28.658369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = gridsearch_ridge_model.best_estimator_.predict(X_train)\nprint(round(rmsle(log_y, preds, True), 4))","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:28.661220Z","iopub.execute_input":"2022-07-23T12:38:28.661915Z","iopub.status.idle":"2022-07-23T12:38:28.674240Z","shell.execute_reply.started":"2022-07-23T12:38:28.661872Z","shell.execute_reply":"2022-07-23T12:38:28.673128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 라쏘 회귀 모델\nfrom sklearn.linear_model import Lasso\nlasso_model = Lasso()\nlasso_alpha = 1 / np.array([0.1,1,2,3,4,10,30,100,200,300,400,800,900,1000])\nlasso_params = {'max_iter' : [3000], 'alpha' : lasso_alpha}\ngridsearch_lasso_model = GridSearchCV(estimator = lasso_model, param_grid = lasso_params, scoring = rmsle_scorer, cv = 5)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:28.675933Z","iopub.execute_input":"2022-07-23T12:38:28.676615Z","iopub.status.idle":"2022-07-23T12:38:28.685772Z","shell.execute_reply.started":"2022-07-23T12:38:28.676570Z","shell.execute_reply":"2022-07-23T12:38:28.684520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_y = np.log(y)\ngridsearch_lasso_model.fit(X_train, log_y)\nprint(gridsearch_lasso_model.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:28.687344Z","iopub.execute_input":"2022-07-23T12:38:28.693226Z","iopub.status.idle":"2022-07-23T12:38:31.563876Z","shell.execute_reply.started":"2022-07-23T12:38:28.693169Z","shell.execute_reply":"2022-07-23T12:38:31.562626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = gridsearch_lasso_model.best_estimator_.predict(X_train)\nprint(round(rmsle(log_y, preds, True),4))","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:38:31.566396Z","iopub.execute_input":"2022-07-23T12:38:31.568135Z","iopub.status.idle":"2022-07-23T12:38:31.589530Z","shell.execute_reply.started":"2022-07-23T12:38:31.568085Z","shell.execute_reply":"2022-07-23T12:38:31.588204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 랜덤포레스트 회귀 모델\nfrom sklearn.ensemble import RandomForestRegressor\nrandomforest_model = RandomForestRegressor()\nrf_params = {'random_state' : [42], 'n_estimators' : [100,120,140]}\ngridsearch_rf_model = GridSearchCV(estimator = randomforest_model, param_grid = rf_params, scoring = rmsle_scorer, cv = 5)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:47:18.154736Z","iopub.execute_input":"2022-07-23T12:47:18.155777Z","iopub.status.idle":"2022-07-23T12:47:18.162505Z","shell.execute_reply.started":"2022-07-23T12:47:18.155723Z","shell.execute_reply":"2022-07-23T12:47:18.161494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_y = np.log(y)\ngridsearch_rf_model.fit(X_train, log_y)\nprint(gridsearch_rf_model.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:47:20.350531Z","iopub.execute_input":"2022-07-23T12:47:20.351076Z","iopub.status.idle":"2022-07-23T12:48:08.114762Z","shell.execute_reply.started":"2022-07-23T12:47:20.351044Z","shell.execute_reply":"2022-07-23T12:48:08.113563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = gridsearch_rf_model.best_estimator_.predict(X_train)\nprint(round(rmsle(log_y, preds, True), 4)) #0.1127, 0.3055","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:44:12.803527Z","iopub.execute_input":"2022-07-23T12:44:12.803916Z","iopub.status.idle":"2022-07-23T12:44:12.896371Z","shell.execute_reply.started":"2022-07-23T12:44:12.803883Z","shell.execute_reply":"2022-07-23T12:44:12.895323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf_pred = gridsearch_rf_model.best_estimator_.predict(X_test)\nsubmission['count'] = np.exp(rf_pred)\nsubmission.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:39:19.919478Z","iopub.execute_input":"2022-07-23T12:39:19.919771Z","iopub.status.idle":"2022-07-23T12:39:20.130858Z","shell.execute_reply.started":"2022-07-23T12:39:19.919744Z","shell.execute_reply":"2022-07-23T12:39:20.129993Z"},"trusted":true},"execution_count":null,"outputs":[]}]}