{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\n# 데이터 경로\ndata_path = '/kaggle/input/bike-sharing-demand/'\n\ntrain = pd.read_csv(data_path + 'train.csv') # train data\ntest = pd.read_csv(data_path + 'test.csv') # test data\nsubmission = pd.read_csv(data_path + 'sampleSubmission.csv') # submission sample data","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-17T12:30:52.439556Z","iopub.execute_input":"2022-07-17T12:30:52.440148Z","iopub.status.idle":"2022-07-17T12:30:52.561136Z","shell.execute_reply.started":"2022-07-17T12:30:52.440087Z","shell.execute_reply":"2022-07-17T12:30:52.559992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape, test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:30:52.563612Z","iopub.execute_input":"2022-07-17T12:30:52.564369Z","iopub.status.idle":"2022-07-17T12:30:52.573245Z","shell.execute_reply.started":"2022-07-17T12:30:52.564326Z","shell.execute_reply":"2022-07-17T12:30:52.571503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()\n\n# datetime : 1시간 간격으로 기록한 일시\n# season : 계절(1,2,3,4 -> 봄, 여름, 가을, 겨울)\n# holiday : 공휴일 여부(1 -> 공휴일)\n# workingday : 근무일 여부 (1 -> 근무일) => 주말과 공휴일이 아니면 근무일이라고 간주함\n# weather : 1->맑음, 2-> 옅은안개or약간 흐림, 3->약간의 눈과 비, 천둥번개, 흐림 4-> 폭우와 천둥번개 ==> 숫자가 클수록 날씨가 안좋음\n# temp : 실제온도\n# atemp : 체감온도\n# humidity : 상대습도\n# casual : 비회원 수\n# registered : 회원 수\n# count :자전거 대여 수량\n","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:30:52.575500Z","iopub.execute_input":"2022-07-17T12:30:52.576352Z","iopub.status.idle":"2022-07-17T12:30:52.606269Z","shell.execute_reply.started":"2022-07-17T12:30:52.576308Z","shell.execute_reply":"2022-07-17T12:30:52.605199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:30:52.611056Z","iopub.execute_input":"2022-07-17T12:30:52.611500Z","iopub.status.idle":"2022-07-17T12:30:52.635616Z","shell.execute_reply.started":"2022-07-17T12:30:52.611458Z","shell.execute_reply":"2022-07-17T12:30:52.634504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:30:52.637474Z","iopub.execute_input":"2022-07-17T12:30:52.638379Z","iopub.status.idle":"2022-07-17T12:30:52.651695Z","shell.execute_reply.started":"2022-07-17T12:30:52.638333Z","shell.execute_reply":"2022-07-17T12:30:52.650415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:30:52.653864Z","iopub.execute_input":"2022-07-17T12:30:52.654777Z","iopub.status.idle":"2022-07-17T12:30:52.681710Z","shell.execute_reply.started":"2022-07-17T12:30:52.654734Z","shell.execute_reply":"2022-07-17T12:30:52.679697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:30:52.684029Z","iopub.execute_input":"2022-07-17T12:30:52.684860Z","iopub.status.idle":"2022-07-17T12:30:52.706654Z","shell.execute_reply.started":"2022-07-17T12:30:52.684814Z","shell.execute_reply":"2022-07-17T12:30:52.704351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature Engineering","metadata":{}},{"cell_type":"code","source":"print(train['datetime'][100]) # datetime 100번째 원소\nprint(train['datetime'][100].split()) # 공백 기준으로 문자열 나누기\nprint(train['datetime'][100].split()[0]) # 날짜\nprint(train['datetime'][100].split()[1]) # 시간","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:30:52.709847Z","iopub.execute_input":"2022-07-17T12:30:52.710498Z","iopub.status.idle":"2022-07-17T12:30:52.725340Z","shell.execute_reply.started":"2022-07-17T12:30:52.710448Z","shell.execute_reply":"2022-07-17T12:30:52.723994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train['datetime'][100].split()[0]) # 날짜\nprint(train['datetime'][100].split()[0].split(\"-\")) # \"-\" 기준으로 문자열 나누기\nprint(train['datetime'][100].split()[0].split(\"-\")[0]) # 연도\nprint(train['datetime'][100].split()[0].split(\"-\")[1]) # 월\nprint(train['datetime'][100].split()[0].split(\"-\")[2]) # 일","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:30:52.727029Z","iopub.execute_input":"2022-07-17T12:30:52.727988Z","iopub.status.idle":"2022-07-17T12:30:52.743007Z","shell.execute_reply.started":"2022-07-17T12:30:52.727939Z","shell.execute_reply":"2022-07-17T12:30:52.741924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train['datetime'][100].split()[1]) # 시간\nprint(train['datetime'][100].split()[1].split(\":\")) # \":\" 기준으로 문자열 나누기\nprint(train['datetime'][100].split()[1].split(\":\")[0]) # 시\nprint(train['datetime'][100].split()[1].split(\":\")[1]) # 분\nprint(train['datetime'][100].split()[1].split(\":\")[2]) # 초","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:30:52.749560Z","iopub.execute_input":"2022-07-17T12:30:52.750567Z","iopub.status.idle":"2022-07-17T12:30:52.762445Z","shell.execute_reply.started":"2022-07-17T12:30:52.750500Z","shell.execute_reply":"2022-07-17T12:30:52.761011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['date'] = train['datetime'].apply(lambda x:x.split()[0]) # 날짜 피처 생성\n\n# 연, 월, 일, 시, 분, 초 피처 차례로 생성\ntrain['year'] = train['datetime'].apply(lambda x:x.split()[0].split(\"-\")[0])\ntrain['month'] = train['datetime'].apply(lambda x:x.split()[0].split(\"-\")[1])\ntrain['day'] = train['datetime'].apply(lambda x:x.split()[0].split(\"-\")[2])\ntrain['hour'] = train['datetime'].apply(lambda x:x.split()[1].split(\":\")[0])\ntrain['minute'] = train['datetime'].apply(lambda x:x.split()[1].split(\":\")[1])\ntrain['second'] = train['datetime'].apply(lambda x:x.split()[1].split(\":\")[2])","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:30:52.764668Z","iopub.execute_input":"2022-07-17T12:30:52.765665Z","iopub.status.idle":"2022-07-17T12:30:52.914435Z","shell.execute_reply.started":"2022-07-17T12:30:52.765621Z","shell.execute_reply":"2022-07-17T12:30:52.913075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from datetime import datetime\nimport calendar\n\nprint(train['date'][100]) # 임의의 날짜\nprint(datetime.strptime(train['date'][100], '%Y-%m-%d')) # datetime 타입으로 변경\n\n# 정수로 요일 반환\nprint(datetime.strptime(train['date'][100], '%Y-%m-%d').weekday())\n\n# 문자열로 요일 반환\nprint(calendar.day_name[datetime.strptime(train['date'][100], '%Y-%m-%d').weekday()])","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:30:52.916440Z","iopub.execute_input":"2022-07-17T12:30:52.917258Z","iopub.status.idle":"2022-07-17T12:30:52.929328Z","shell.execute_reply.started":"2022-07-17T12:30:52.917215Z","shell.execute_reply":"2022-07-17T12:30:52.927987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['weekday'] = train['date'].apply(lambda dateString:\n                                      calendar.day_name[datetime.strptime(dateString, \"%Y-%m-%d\").weekday()])","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:30:52.931334Z","iopub.execute_input":"2022-07-17T12:30:52.931755Z","iopub.status.idle":"2022-07-17T12:30:53.200578Z","shell.execute_reply.started":"2022-07-17T12:30:52.931715Z","shell.execute_reply":"2022-07-17T12:30:53.199321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['season'] = train['season'].map({1: 'Spring',\n                                       2: 'Summer',\n                                       3: 'Fall',\n                                       4: 'Winter'})\ntrain['weather'] = train['weather'].map({1: 'Clear',\n                                         2: 'Mist, Few clouds',\n                                         3: 'Light snow, Rain, Thunderstrom',\n                                         4: 'Heavy Rain, Thunderstorm, Snow, Fog'})","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:30:53.202147Z","iopub.execute_input":"2022-07-17T12:30:53.202462Z","iopub.status.idle":"2022-07-17T12:30:53.215766Z","shell.execute_reply.started":"2022-07-17T12:30:53.202434Z","shell.execute_reply":"2022-07-17T12:30:53.214221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:30:53.217583Z","iopub.execute_input":"2022-07-17T12:30:53.218177Z","iopub.status.idle":"2022-07-17T12:30:53.251189Z","shell.execute_reply.started":"2022-07-17T12:30:53.218127Z","shell.execute_reply":"2022-07-17T12:30:53.249412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* 3달씩 묶으면 계절\n* date -> year, month, day","metadata":{}},{"cell_type":"markdown","source":"### Visualization","metadata":{}},{"cell_type":"code","source":" import seaborn as sns\n import matplotlib as mpl\n import matplotlib.pyplot as plt\n %matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:30:53.254003Z","iopub.execute_input":"2022-07-17T12:30:53.254361Z","iopub.status.idle":"2022-07-17T12:30:53.269705Z","shell.execute_reply.started":"2022-07-17T12:30:53.254327Z","shell.execute_reply":"2022-07-17T12:30:53.268378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Distribution plot (분포도)","metadata":{}},{"cell_type":"code","source":"mpl.rc('font', size = 15) # 폰트 크기 15로 출력 셋팅\nsns.displot(train['count']) # 분포도 출력","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:30:53.273300Z","iopub.execute_input":"2022-07-17T12:30:53.274565Z","iopub.status.idle":"2022-07-17T12:30:53.694926Z","shell.execute_reply.started":"2022-07-17T12:30:53.274507Z","shell.execute_reply":"2022-07-17T12:30:53.693464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"회귀 모델이 좋은 성능을 내려면 정규분포를 따라야하는데 그렇지 않음..  \n로그변환을 해줌. 왼쪽 편향되어있을 때 사용함","metadata":{}},{"cell_type":"code","source":"sns.displot(np.log(train['count']))","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:30:53.697198Z","iopub.execute_input":"2022-07-17T12:30:53.697686Z","iopub.status.idle":"2022-07-17T12:30:54.079809Z","shell.execute_reply.started":"2022-07-17T12:30:53.697637Z","shell.execute_reply":"2022-07-17T12:30:54.078167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Bar plot (막대 그래프)","metadata":{}},{"cell_type":"code","source":"# step1. (m x n) Figure 준비\nmpl.rc('font', size = 14) # 폰트크기 설정\nmpl.rc('axes', titlesize=15) # 각 축의 제목 크기 설정\nfigure, axes = plt.subplots(nrows=3, ncols=2) # 3행 2열 figure 생성\nplt.tight_layout() # 그래프 사이에 여백 확보\nfigure.set_size_inches(10, 9) # 전체 Figure 크기를 10x9인치로 설정\n\n# step2. 각 축에 서브플롯 할당\nsns.barplot(x='year', y='count', data=train, ax=axes[0, 0])\nsns.barplot(x='month', y='count', data=train, ax=axes[0, 1])\nsns.barplot(x='day', y='count', data=train, ax=axes[1, 0])\nsns.barplot(x='hour', y='count', data=train, ax=axes[1, 1])\nsns.barplot(x='minute', y='count', data=train, ax=axes[2, 0])\nsns.barplot(x='second', y='count', data=train, ax=axes[2, 1])\n\n# step3. 세부설정\naxes[0, 0].set(title='Rental amounts by year')\naxes[0, 1].set(title='Rental amounts by month')\naxes[1, 0].set(title='Rental amounts by day')\naxes[1, 1].set(title='Rental amounts by hour')\naxes[2, 0].set(title='Rental amounts by minute')\naxes[2, 1].set(title='Rental amounts by second')\n# 1행의 x축 라벨 2개를 90도 회전 (보기 편하도록)\naxes[1, 0].tick_params(axis='x', labelrotation=90)\naxes[1, 1].tick_params(axis='x', labelrotation=90)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:30:54.081812Z","iopub.execute_input":"2022-07-17T12:30:54.082864Z","iopub.status.idle":"2022-07-17T12:30:57.856144Z","shell.execute_reply.started":"2022-07-17T12:30:54.082814Z","shell.execute_reply":"2022-07-17T12:30:57.854610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Box plot(박스플롯)  \n* 범주형 데이터에 따른 수치형 데이터 정보를 나타내는 그래프","metadata":{}},{"cell_type":"code","source":"# Step1. m행 n형 Figure 준비\nfigure, axes = plt.subplots(nrows=2, ncols=2) # 2행 2열\nplt.tight_layout()\nfigure.set_size_inches(10, 10)\n\n# Step2. 서브 플롯 할당\n# 계절, 날씨, 공휴일, 근무일별 대여 수량 박스플롯\nsns.boxplot(x='season', y='count', data=train, ax=axes[0, 0])\nsns.boxplot(x='weather',y='count', data=train, ax=axes[0, 1])\nsns.boxplot(x='holiday',y='count', data=train, ax=axes[1, 0])\nsns.boxplot(x='workingday', y='count', data=train, ax=axes[1, 1])\n\n# Step3. 세부 설정\naxes[0, 0].set(title='Box Plot On Count Across Season')\naxes[0, 1].set(title='Box Plot On Count Across Weather')\naxes[1, 0].set(title='Box Plot On Count Across Holiday')\naxes[1, 1].set(title='Box Plot On Count Across Workingday')\n\naxes[0, 1].tick_params(axis='x', labelrotation=10)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:30:57.857447Z","iopub.execute_input":"2022-07-17T12:30:57.857772Z","iopub.status.idle":"2022-07-17T12:30:58.548714Z","shell.execute_reply.started":"2022-07-17T12:30:57.857743Z","shell.execute_reply":"2022-07-17T12:30:58.547313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Point plot(포인트 플롯)","metadata":{}},{"cell_type":"code","source":"# Step1. m행 n열 Figure 준비\nmpl.rc('font', size=11)\nfigure, axes = plt.subplots(nrows=5) # 5행 1열\nfigure.set_size_inches(12, 18)\n\n# Step2. 서브플롯 할당\n# 근무일, 공휴일, 요일, 계절, 날씨에 따른 시간대별 평균 대여 수량 포인트플롯\nsns.pointplot(x='hour', y='count', data=train, hue = 'workingday', ax=axes[0])\nsns.pointplot(x='hour', y='count', data=train, hue = 'holiday', ax=axes[1])\nsns.pointplot(x='hour', y='count', data=train, hue = 'weekday', ax=axes[2])\nsns.pointplot(x='hour', y='count', data=train, hue = 'season', ax=axes[3])\nsns.pointplot(x='hour', y='count', data=train, hue = 'weather', ax=axes[4])","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:30:58.551013Z","iopub.execute_input":"2022-07-17T12:30:58.551460Z","iopub.status.idle":"2022-07-17T12:31:15.157546Z","shell.execute_reply.started":"2022-07-17T12:30:58.551417Z","shell.execute_reply":"2022-07-17T12:31:15.156074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Scatter plot graph with regression line (회귀선을 포함한 산점도 그래프)","metadata":{}},{"cell_type":"code","source":"# Step1. m행 n열 Figure 준비\nmpl.rc('font', size=15)\nfigure, axes = plt.subplots(nrows=2, ncols=2)\nplt.tight_layout()\nfigure.set_size_inches(7, 6)\n\n# Step2. 서브플롯 할당\n# 온도, 체감 온도, 풍속, 습도 별 대여 수량 산점도 그래프\nsns.regplot(x='temp', y='count', data=train, ax=axes[0, 0],\n           scatter_kws={'alpha':0.2}, line_kws={'color':'blue'})\nsns.regplot(x='atemp', y='count', data=train, ax=axes[0, 1],\n           scatter_kws={'alpha':0.2}, line_kws={'color':'blue'})\nsns.regplot(x='windspeed', y='count', data=train, ax=axes[1, 0],\n           scatter_kws={'alpha':0.2}, line_kws={'color':'blue'})\nsns.regplot(x='humidity', y='count', data=train, ax=axes[1, 1],\n           scatter_kws={'alpha':0.2}, line_kws={'color':'blue'})","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:31:15.158872Z","iopub.execute_input":"2022-07-17T12:31:15.159766Z","iopub.status.idle":"2022-07-17T12:31:18.738587Z","shell.execute_reply.started":"2022-07-17T12:31:15.159727Z","shell.execute_reply":"2022-07-17T12:31:18.737196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Heatmap (히트맵)\n* 수치형 데이터간의 어떠한 상관관계가 있는지에 대해 알아봄","metadata":{}},{"cell_type":"code","source":"train[['temp','atemp','humidity','windspeed','count']].corr()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:31:18.740439Z","iopub.execute_input":"2022-07-17T12:31:18.740899Z","iopub.status.idle":"2022-07-17T12:31:18.768478Z","shell.execute_reply.started":"2022-07-17T12:31:18.740857Z","shell.execute_reply":"2022-07-17T12:31:18.766999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Feature 간의 상관관계 매트릭스\ncorrMatrix = train[['temp','atemp','humidity','windspeed','count']].corr()\nfig, ax = plt.subplots()\nfig.set_size_inches(10, 10)\nsns.heatmap(corrMatrix, annot=True) # 상관관계 히트맵 그리기\nax.set(title='Heatmap of Numerical Data')","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:31:18.770624Z","iopub.execute_input":"2022-07-17T12:31:18.771084Z","iopub.status.idle":"2022-07-17T12:31:19.118460Z","shell.execute_reply.started":"2022-07-17T12:31:18.771038Z","shell.execute_reply":"2022-07-17T12:31:19.117464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Baseline Model","metadata":{}},{"cell_type":"markdown","source":"EDA완료 이후 실제 모델 제작을 위해 새로 데이터 불러와서 시작","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# 데이터 경로\ndata_path = '/kaggle/input/bike-sharing-demand/'\n\ntrain = pd.read_csv(data_path + 'train.csv') # train data\ntest = pd.read_csv(data_path + 'test.csv') # test data\nsubmission = pd.read_csv(data_path + 'sampleSubmission.csv') # submission sample data","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:31:19.120057Z","iopub.execute_input":"2022-07-17T12:31:19.120788Z","iopub.status.idle":"2022-07-17T12:31:19.172738Z","shell.execute_reply.started":"2022-07-17T12:31:19.120752Z","shell.execute_reply":"2022-07-17T12:31:19.171063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* point plot에서 weather가 4인 데이터는 이상치이므로 제거","metadata":{}},{"cell_type":"code","source":"# 훈련 데이터에서 weather가 4가 아닌 데이터만 추출\ntrain = train[train['weather'] != 4]","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:31:19.174323Z","iopub.execute_input":"2022-07-17T12:31:19.175315Z","iopub.status.idle":"2022-07-17T12:31:19.182055Z","shell.execute_reply.started":"2022-07-17T12:31:19.175280Z","shell.execute_reply":"2022-07-17T12:31:19.181146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data_temp = pd.concat([train, test])\nall_data_temp","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:31:19.183336Z","iopub.execute_input":"2022-07-17T12:31:19.184589Z","iopub.status.idle":"2022-07-17T12:31:19.222851Z","shell.execute_reply.started":"2022-07-17T12:31:19.184519Z","shell.execute_reply":"2022-07-17T12:31:19.221118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"index가 train, test에 붙어있는게 그대로 붙어서 합쳐져서 index값이 이상하게 보임  \n이를 위해 ignore_index=True 처리","metadata":{}},{"cell_type":"code","source":"all_data = pd.concat([train, test], ignore_index=True)\nall_data","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:31:19.229852Z","iopub.execute_input":"2022-07-17T12:31:19.230294Z","iopub.status.idle":"2022-07-17T12:31:19.264113Z","shell.execute_reply.started":"2022-07-17T12:31:19.230261Z","shell.execute_reply":"2022-07-17T12:31:19.262475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from datetime import datetime\n\n# 날짜 피처 생성\nall_data['date'] = all_data['datetime'].apply(lambda x:x.split()[0])\n# 연도 피처 생성\nall_data['year'] = all_data['datetime'].apply(lambda x:x.split()[0].split(\"-\")[0])\n# 월 피처 생성\nall_data['month'] = all_data['datetime'].apply(lambda x:x.split()[0].split(\"-\")[1])\n# 시 피처 생성\nall_data['hour'] = all_data['datetime'].apply(lambda x:x.split()[1].split(\":\")[0])\n# 요일 피처 생성\nall_data['weekday'] = all_data['date'].apply(lambda dateString : datetime.strptime(dateString, \"%Y-%m-%d\").weekday()) # 요일을 정수형으로 생성\n\n# train은 1~19일 test는 20 ~ 말일 이므로 사용할 필요가 없음","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:31:19.265832Z","iopub.execute_input":"2022-07-17T12:31:19.266244Z","iopub.status.idle":"2022-07-17T12:31:19.571698Z","shell.execute_reply.started":"2022-07-17T12:31:19.266208Z","shell.execute_reply":"2022-07-17T12:31:19.570378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### 쓸모없는 Feature 제거\n* season에 month에 관한 내용이 담김 (3개월단위 = season)\n* windspeed 상관관계 약함 등등","metadata":{}},{"cell_type":"code","source":"drop_features = ['casual', 'registered', 'datetime', 'date', 'windspeed', 'month']\nall_data = all_data.drop(drop_features, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:31:19.573475Z","iopub.execute_input":"2022-07-17T12:31:19.573816Z","iopub.status.idle":"2022-07-17T12:31:19.589930Z","shell.execute_reply.started":"2022-07-17T12:31:19.573786Z","shell.execute_reply":"2022-07-17T12:31:19.588524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 데이터 나누기\n* Feature 작업 완료했으니 다시 train, test로 나누기","metadata":{}},{"cell_type":"code","source":"X_train = all_data[~pd.isnull(all_data['count'])]\nX_test = all_data[pd.isnull(all_data['count'])]","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:31:19.591937Z","iopub.execute_input":"2022-07-17T12:31:19.592591Z","iopub.status.idle":"2022-07-17T12:31:19.601109Z","shell.execute_reply.started":"2022-07-17T12:31:19.592556Z","shell.execute_reply":"2022-07-17T12:31:19.599801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 타켓의 count 제거\nX_train = X_train.drop(['count'], axis=1)\nX_test = X_test.drop(['count'], axis=1)\n\ny = train['count'] # 타켓값","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:31:19.602546Z","iopub.execute_input":"2022-07-17T12:31:19.602967Z","iopub.status.idle":"2022-07-17T12:31:19.615191Z","shell.execute_reply.started":"2022-07-17T12:31:19.602924Z","shell.execute_reply":"2022-07-17T12:31:19.614158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:31:19.616553Z","iopub.execute_input":"2022-07-17T12:31:19.616979Z","iopub.status.idle":"2022-07-17T12:31:19.637431Z","shell.execute_reply.started":"2022-07-17T12:31:19.616878Z","shell.execute_reply":"2022-07-17T12:31:19.636102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 평가지표인 RMSLE 계산하는 함수 작성","metadata":{}},{"cell_type":"code","source":"import numpy as np\ndef rmsle(y_true, y_pred, convertExp=True):\n    # 지수변환 (count를 로그변환해서 예측하기 때문에 다시 돌려줘야함)\n    if convertExp:\n        y_true = np.exp(y_true)\n        y_pred = np.exp(y_pred)\n    \n    # 로그변환 후 결측값을 0으로 변환\n    log_true = np.nan_to_num(np.log(y_true+1))\n    log_pred = np.nan_to_num(np.log(y_pred+1))\n    \n    # RMSLE 계산\n    output = np.sqrt(np.mean((log_true - log_pred)**2)) # RMSLE 공식을 그래서 작성\n    return output","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:31:19.639634Z","iopub.execute_input":"2022-07-17T12:31:19.640073Z","iopub.status.idle":"2022-07-17T12:31:19.650178Z","shell.execute_reply.started":"2022-07-17T12:31:19.640033Z","shell.execute_reply":"2022-07-17T12:31:19.648195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 모델 훈련","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\n\nlinear_reg_model = LinearRegression()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:31:19.652014Z","iopub.execute_input":"2022-07-17T12:31:19.652699Z","iopub.status.idle":"2022-07-17T12:31:19.666682Z","shell.execute_reply.started":"2022-07-17T12:31:19.652659Z","shell.execute_reply":"2022-07-17T12:31:19.665568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_y = np.log(y) # 타겟값 로그변환\nlinear_reg_model.fit(X_train, log_y) # 모델 훈련","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:31:19.668287Z","iopub.execute_input":"2022-07-17T12:31:19.668647Z","iopub.status.idle":"2022-07-17T12:31:19.714990Z","shell.execute_reply.started":"2022-07-17T12:31:19.668618Z","shell.execute_reply":"2022-07-17T12:31:19.713287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 모델 성능 검증\n* 예시를 들어서 검증하는 것..학습데이터로 검증하는것은 안됨","metadata":{}},{"cell_type":"code","source":"preds = linear_reg_model.predict(X_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:31:19.717407Z","iopub.execute_input":"2022-07-17T12:31:19.718228Z","iopub.status.idle":"2022-07-17T12:31:19.775603Z","shell.execute_reply.started":"2022-07-17T12:31:19.718175Z","shell.execute_reply":"2022-07-17T12:31:19.773128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"선형 회귀의 RMSLE의 값 : {rmsle(log_y, preds, True):.4f}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:31:19.779332Z","iopub.execute_input":"2022-07-17T12:31:19.782446Z","iopub.status.idle":"2022-07-17T12:31:19.805497Z","shell.execute_reply.started":"2022-07-17T12:31:19.782387Z","shell.execute_reply":"2022-07-17T12:31:19.803688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 예측 및 결과 제출","metadata":{}},{"cell_type":"code","source":"linearreg_preds = linear_reg_model.predict(X_test) # 테스트 데이터로 예측\n\nsubmission['count'] = np.exp(linearreg_preds)  # 지수변환\nsubmission.to_csv('submission.csv', index=False) # 파일로 저장","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:31:19.808108Z","iopub.execute_input":"2022-07-17T12:31:19.809860Z","iopub.status.idle":"2022-07-17T12:31:19.902002Z","shell.execute_reply.started":"2022-07-17T12:31:19.809810Z","shell.execute_reply":"2022-07-17T12:31:19.900441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 성능개선\n### 릿지(Ridge) 회귀 모델\n* 규제(regularization)을 통해 과대적합(overfitting)을 방지를 해주는 모델","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import Ridge\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn import metrics\n\nridge_model = Ridge()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:36:18.849281Z","iopub.execute_input":"2022-07-17T12:36:18.849779Z","iopub.status.idle":"2022-07-17T12:36:18.857293Z","shell.execute_reply.started":"2022-07-17T12:36:18.849742Z","shell.execute_reply":"2022-07-17T12:36:18.855984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 하이퍼파라미터 값 목록\nridge_params = {'max_iter':[3000], 'alpha':[0.1, 1, 2, 3, 4, 10, 30, 100, 200, 300, 400, 800, 900, 1000]}\n\n# 교차 검증용 평가 함수(RMSLE 점수 계산)\nrmsle_scorer = metrics.make_scorer(rmsle, greater_is_better=False)\n\n# 그리드서치(with 릿지) 객체 생성\ngridsearch_ridge_model = GridSearchCV(estimator = ridge_model,   # 릿지모델\n                                      param_grid = ridge_params, # 값 목록\n                                      scoring=rmsle_scorer,      # 평가지표\n                                      cv=5)                      # 교차 검증 분할 수","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:41:18.102069Z","iopub.execute_input":"2022-07-17T12:41:18.102594Z","iopub.status.idle":"2022-07-17T12:41:18.110901Z","shell.execute_reply.started":"2022-07-17T12:41:18.102557Z","shell.execute_reply":"2022-07-17T12:41:18.109982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"그리드서치 수행","metadata":{}},{"cell_type":"code","source":"log_y = np.log(y) # 타깃값 로그변환\ngridsearch_ridge_model.fit(X_train, log_y) # 훈련 (그리드서치)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:44:12.214553Z","iopub.execute_input":"2022-07-17T12:44:12.215080Z","iopub.status.idle":"2022-07-17T12:44:15.147797Z","shell.execute_reply.started":"2022-07-17T12:44:12.215046Z","shell.execute_reply":"2022-07-17T12:44:15.145321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('최적 하이퍼파라미터 : ', gridsearch_ridge_model.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:45:04.063285Z","iopub.execute_input":"2022-07-17T12:45:04.063878Z","iopub.status.idle":"2022-07-17T12:45:04.070781Z","shell.execute_reply.started":"2022-07-17T12:45:04.063836Z","shell.execute_reply":"2022-07-17T12:45:04.069799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 성능 검증","metadata":{}},{"cell_type":"code","source":"# 예측\npreds = gridsearch_ridge_model.best_estimator_.predict(X_train)\n# 평가\nprint(f\"릿지 회귀 RMSLE 값 : {rmsle(log_y, preds, True):.4f}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:46:52.313616Z","iopub.execute_input":"2022-07-17T12:46:52.314123Z","iopub.status.idle":"2022-07-17T12:46:52.344724Z","shell.execute_reply.started":"2022-07-17T12:46:52.314088Z","shell.execute_reply":"2022-07-17T12:46:52.343257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 라쏘(Lasso) 회귀 모델","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import Lasso\n\n# 모델 생성\nlasso_model = Lasso()\n# 하이퍼파라미터 값 목록\nlasso_alpha = 1/np.array([0.1, 1, 2, 3, 4, 10, 30, 100, 200, 300, 400, 800, 900, 1000])\nlasso_params = {'max_iter':[3000], 'alpha':lasso_alpha}\n\n# 그리드서치(with 라쏘) 객체 생성\ngridsearch_lasso_model = GridSearchCV(estimator=lasso_model,\n                                      param_grid=lasso_params,\n                                      scoring=rmsle_scorer,\n                                      cv=5)\n\n# 그리드 서치 수행\nlog_y = np.log(y)\ngridsearch_lasso_model.fit(X_train, log_y)\n\nprint('최적 하이퍼파라미터 : ', gridsearch_lasso_model.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:02:33.350200Z","iopub.execute_input":"2022-07-17T13:02:33.350673Z","iopub.status.idle":"2022-07-17T13:02:39.991854Z","shell.execute_reply.started":"2022-07-17T13:02:33.350637Z","shell.execute_reply":"2022-07-17T13:02:39.990396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 성능검증","metadata":{}},{"cell_type":"code","source":"# 예측\npreds = gridsearch_lasso_model.best_estimator_.predict(X_train)\n\n# 평가\nprint(f\"라쏘 회귀 RMSLE 값 : {rmsle(log_y, preds, True):.4f}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:08:56.822457Z","iopub.execute_input":"2022-07-17T13:08:56.823187Z","iopub.status.idle":"2022-07-17T13:08:56.858090Z","shell.execute_reply.started":"2022-07-17T13:08:56.823150Z","shell.execute_reply":"2022-07-17T13:08:56.856692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## RandomForest (랜덤포레스트) 회귀 모델","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\n\n# 모델 생성\nrandomforest_model = RandomForestRegressor()\n# 그리드서치 객체 생성\nrf_params = {'random_state':[42], 'n_estimators':[100, 120, 140]}\ngridsearch_random_forest_model = GridSearchCV(estimator=randomforest_model,\n                                              param_grid=rf_params,\n                                              scoring=rmsle_scorer,\n                                              cv=5)\n\n# 그리드서치 수행\nlog_y = np.log(y)\ngridsearch_random_forest_model.fit(X_train, log_y)\nprint(\"최적 하이퍼파라미터 : \", gridsearch_random_forest_model.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:24:45.490752Z","iopub.execute_input":"2022-07-17T13:24:45.491365Z","iopub.status.idle":"2022-07-17T13:25:36.595347Z","shell.execute_reply.started":"2022-07-17T13:24:45.491318Z","shell.execute_reply":"2022-07-17T13:25:36.593904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 모델 성능 검증","metadata":{}},{"cell_type":"code","source":"# 예측\npreds = gridsearch_random_forest_model.best_estimator_.predict(X_train)\n\n# 평가\nprint(f\"랜덤 포레스트 회귀 RMSLE 값 : {rmsle(log_y, preds, True):.4f}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:27:05.453227Z","iopub.execute_input":"2022-07-17T13:27:05.454785Z","iopub.status.idle":"2022-07-17T13:27:05.815448Z","shell.execute_reply.started":"2022-07-17T13:27:05.454717Z","shell.execute_reply":"2022-07-17T13:27:05.814111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nrandomforest_preds = gridsearch_random_forest_model.best_estimator_.predict(X_test)\n\nfigure, axes = plt.subplots(ncols=2)\nfigure.set_size_inches(10, 4)\n\nsns.histplot(y, bins=50, ax=axes[0])\naxes[0].set_title('Train Data Distribution')\nsns.histplot(np.exp(randomforest_preds), bins=50, ax=axes[1])\naxes[1].set_title('Predicted Test Data Distribution')","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:53:19.294723Z","iopub.execute_input":"2022-07-17T13:53:19.295173Z","iopub.status.idle":"2022-07-17T13:53:20.035336Z","shell.execute_reply.started":"2022-07-17T13:53:19.295140Z","shell.execute_reply":"2022-07-17T13:53:20.034122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['count'] = np.exp(randomforest_preds) # 지수변환\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:54:23.215472Z","iopub.execute_input":"2022-07-17T13:54:23.215978Z","iopub.status.idle":"2022-07-17T13:54:23.253297Z","shell.execute_reply.started":"2022-07-17T13:54:23.215928Z","shell.execute_reply":"2022-07-17T13:54:23.252217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}