{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 6장 자전거 대여 수요 예측 경진대회 환경 세팅된 노트북 양식","metadata":{"papermill":{"duration":0.026639,"end_time":"2021-08-16T04:01:01.662249","exception":false,"start_time":"2021-08-16T04:01:01.63561","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"> # **데이터 둘러보기**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\ndata_path = '/kaggle/input/bike-sharing-demand/'\n\ntrain = pd.read_csv(data_path + 'train.csv')\ntest = pd.read_csv(data_path + 'test.csv')\nsubmission = pd.read_csv(data_path + 'sampleSubmission.csv')\n\n# 훈련 데이터와, 테스트 데이터 크기 확인\ntrain.shape, test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:42:32.351711Z","iopub.execute_input":"2022-07-23T12:42:32.352036Z","iopub.status.idle":"2022-07-23T12:42:32.400652Z","shell.execute_reply.started":"2022-07-23T12:42:32.351998Z","shell.execute_reply":"2022-07-23T12:42:32.399824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 두 데이터의 피처 개수(열의 개수)가 다르므로 데이터 직접 확인 시도\n# head 함수는 첫 5행을 출력함\ntrain.head()\ntest.head()\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:42:32.402256Z","iopub.execute_input":"2022-07-23T12:42:32.403878Z","iopub.status.idle":"2022-07-23T12:42:32.413772Z","shell.execute_reply.started":"2022-07-23T12:42:32.403841Z","shell.execute_reply":"2022-07-23T12:42:32.412821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* 두 데이터의 차이는 casual, registered, count 피처\n* test 데이터에 없는 피처는 제거","metadata":{}},{"cell_type":"code","source":"#info() 함수를 이용하면 결측값, 데이터 타입 파악 가능\ntrain.info()\ntest.info()\n#두 데이터 모두 결측값 없는 것 확인 완료 \n#결측값이 있었다면 평균값, 중앙값, 최빈값 등으로 대체 or 결측값을 예측 or 결측값 포함하는 피처 아예 제거","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:42:32.415439Z","iopub.execute_input":"2022-07-23T12:42:32.416100Z","iopub.status.idle":"2022-07-23T12:42:32.444499Z","shell.execute_reply.started":"2022-07-23T12:42:32.416056Z","shell.execute_reply":"2022-07-23T12:42:32.442496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> # **피처 엔지니어링**  \n데이터 시각화 이전에 시각화에 적합하게 피처 변환","metadata":{}},{"cell_type":"code","source":"# datetime 피처가 object 타입이므로 시각화에 적합하지 않음\n# datetime 피처를 split() 함수를 이용해서 연도,월,일,시간,분,초로 나누기\n# 먼저 100번째 원소를 예시로 연습\n\nprint(train['datetime'][100])\nprint(train['datetime'][100].split()) # 공백을 기준으로 문자열 나누기\nprint(train['datetime'][100].split()[0]) # 날짜\nprint(train['datetime'][100].split()[1]) # 시간\n\nprint(train['datetime'][100].split()[0].split(\"-\")) \nprint(train['datetime'][100].split()[0].split(\"-\")[0]) # 연도\nprint(train['datetime'][100].split()[0].split(\"-\")[1]) # 월\nprint(train['datetime'][100].split()[0].split(\"-\")[2]) # 일\n\nprint(train['datetime'][100].split()[1].split(\":\"))\nprint(train['datetime'][100].split()[1].split(\":\")[0]) # 시간\nprint(train['datetime'][100].split()[1].split(\":\")[1]) # 분\nprint(train['datetime'][100].split()[1].split(\":\")[2]) # 초","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:42:32.445886Z","iopub.execute_input":"2022-07-23T12:42:32.446130Z","iopub.status.idle":"2022-07-23T12:42:32.460911Z","shell.execute_reply.started":"2022-07-23T12:42:32.446102Z","shell.execute_reply":"2022-07-23T12:42:32.460083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 판다스 apply() 함수로 앞 부분의 연습 로직을 적용해서 '파생 피처' 생성하기\n# apply() 함수는 map() 함수의 2차원 버전..\n# lambda x \n\ntrain['date'] = train['datetime'].apply(lambda x: x.split()[0])\n\ntrain['year'] = train['datetime'].apply(lambda x: x.split()[0].split(\"-\")[0])\ntrain['month'] = train['datetime'].apply(lambda x: x.split()[0].split(\"-\")[1])\ntrain['day'] = train['datetime'].apply(lambda x: x.split()[0].split(\"-\")[2])\n\ntrain['hour'] = train['datetime'].apply(lambda x: x.split()[1].split(\":\")[0])\ntrain['minute'] = train['datetime'].apply(lambda x: x.split()[1].split(\":\")[1])\ntrain['second'] = train['datetime'].apply(lambda x: x.split()[1].split(\":\")[2])","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:42:32.462839Z","iopub.execute_input":"2022-07-23T12:42:32.463630Z","iopub.status.idle":"2022-07-23T12:42:32.537720Z","shell.execute_reply.started":"2022-07-23T12:42:32.463588Z","shell.execute_reply":"2022-07-23T12:42:32.536803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 요일 피처 생성을 위해 calendar, datetime 라이브러리 불러오기\nfrom datetime import datetime\nimport calendar\n\n# 원리 이해를 위해 100번째 날짜로 예시\nprint(train['date'][100])\nprint(datetime.strptime(train['date'][100], '%Y-%m-%d')) # datetime으로 변경\nprint(datetime.strptime(train['date'][100], '%Y-%m-%d').weekday()) # 정수로 요일 반환\nprint(calendar.day_name[datetime.strptime(train['date'][100], '%Y-%m-%d').weekday()]) # 문자열로 요일 반환\n\n# 원리 이해 바탕으로 요일(weekday) 피처 생성\ntrain['weekday'] = train['date'].apply(\n    lambda dataString :\n    calendar.day_name[datetime.strptime(dataString,'%Y-%m-%d').weekday()])","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:42:32.539461Z","iopub.execute_input":"2022-07-23T12:42:32.539695Z","iopub.status.idle":"2022-07-23T12:42:32.735830Z","shell.execute_reply.started":"2022-07-23T12:42:32.539669Z","shell.execute_reply":"2022-07-23T12:42:32.734978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **strptime 함수 이해 안됨!!!!!!!!!!!!!!!!!!!!!!**\n# **dataString 이해 안됨!!!!!!**","metadata":{}},{"cell_type":"code","source":"# season 피처, weather 피처는 숫자로 표현되어 있어 의미 파악 힘듬\ntrain['season']=train['season'].map({1:'Spring',\n                                     2:'Summer',\n                                     3:'Fall',\n                                     4:'Winter'})\ntrain['weather']=train['weather'].map({1:'Clear',\n                                       2:'Mist, Few clouds',\n                                       3:'Light Snow, Rain, Thunderstorm',\n                                       4:'Heavy Rain, Thunderstorm, Snow, Fog'})\ntrain.head()\n#변화 확인\n#date, year~ weekday 피처 추가 되었음\n#weather, season 피처 값 문자로 변환\n\n#date에 있는 정보는 뒤에 중복되므로 제거 예정\n#month로 season으로 대체 가능하므로 제거 예정","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:42:32.737139Z","iopub.execute_input":"2022-07-23T12:42:32.737481Z","iopub.status.idle":"2022-07-23T12:42:32.766703Z","shell.execute_reply.started":"2022-07-23T12:42:32.737441Z","shell.execute_reply":"2022-07-23T12:42:32.765925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> # **데이터 시각화**\nmatplotlib, seaborn 라이브러리 활용","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:42:32.768413Z","iopub.execute_input":"2022-07-23T12:42:32.768619Z","iopub.status.idle":"2022-07-23T12:42:32.773532Z","shell.execute_reply.started":"2022-07-23T12:42:32.768595Z","shell.execute_reply":"2022-07-23T12:42:32.772966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 수치형 데이터의 집계 값을 알려주는 분포도-displot을 사용해서 타깃값인 count의 분포도 확인\n# 훈련시 타깃값을 그대로 사용할지 변환해 사용할지 파악할 수 있기 때문\nmpl.rc('font', size=15)\nsns.displot(train['count'])","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:42:32.774478Z","iopub.execute_input":"2022-07-23T12:42:32.774850Z","iopub.status.idle":"2022-07-23T12:42:33.210479Z","shell.execute_reply.started":"2022-07-23T12:42:32.774819Z","shell.execute_reply":"2022-07-23T12:42:33.209684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#회귀 모델이 좋은 성능을 내려면 정규분포를 따라야하는데, 현재 왼쪽으로 치우쳐져있음\n#왼쪽으로 치우친 데이터는 로그변환을 통해 바꾸어줌\nsns.displot(np.log(train['count']))\n#log를 취한 값이 정규분포에 가까우므로 이 값을 사용함. 대신 후 처리로 다시 지수변환을 해주어야함","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:42:33.211550Z","iopub.execute_input":"2022-07-23T12:42:33.211765Z","iopub.status.idle":"2022-07-23T12:42:33.619153Z","shell.execute_reply.started":"2022-07-23T12:42:33.211740Z","shell.execute_reply":"2022-07-23T12:42:33.618316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 연도,월,일,시,분,초는 범주형 데이터\n# 막대그래프로 평균값 확인해보기\nsns.barplot(x='year', y='count', data= train)\n#2012년이 대여가 많다","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:42:33.620525Z","iopub.execute_input":"2022-07-23T12:42:33.620954Z","iopub.status.idle":"2022-07-23T12:42:33.939863Z","shell.execute_reply.started":"2022-07-23T12:42:33.620924Z","shell.execute_reply":"2022-07-23T12:42:33.939070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.barplot(x='month', y='count', data= train)\n#여름에 대여가 많다","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:42:33.941546Z","iopub.execute_input":"2022-07-23T12:42:33.941848Z","iopub.status.idle":"2022-07-23T12:42:34.664770Z","shell.execute_reply.started":"2022-07-23T12:42:33.941809Z","shell.execute_reply":"2022-07-23T12:42:34.663804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.barplot(x='day', y='count', data= train)\n#일별 대여 차이는 뚜렷하지 않고, 20일 이후 데이터가 없으므로 사용 불가능","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:42:34.666309Z","iopub.execute_input":"2022-07-23T12:42:34.666596Z","iopub.status.idle":"2022-07-23T12:42:35.676353Z","shell.execute_reply.started":"2022-07-23T12:42:34.666566Z","shell.execute_reply":"2022-07-23T12:42:35.675484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.barplot(x='hour', y='count', data= train)\n#8시, 17~18시 같이0 출최근 시간에 사람이 많다.","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:42:35.678810Z","iopub.execute_input":"2022-07-23T12:42:35.679031Z","iopub.status.idle":"2022-07-23T12:42:36.893477Z","shell.execute_reply.started":"2022-07-23T12:42:35.679004Z","shell.execute_reply":"2022-07-23T12:42:36.892600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.barplot(x='minute', y='count', data= train)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:42:36.894753Z","iopub.execute_input":"2022-07-23T12:42:36.894979Z","iopub.status.idle":"2022-07-23T12:42:37.165932Z","shell.execute_reply.started":"2022-07-23T12:42:36.894953Z","shell.execute_reply":"2022-07-23T12:42:37.164918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.barplot(x='second', y='count', data= train)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:42:37.167096Z","iopub.execute_input":"2022-07-23T12:42:37.167339Z","iopub.status.idle":"2022-07-23T12:42:37.479251Z","shell.execute_reply.started":"2022-07-23T12:42:37.167309Z","shell.execute_reply":"2022-07-23T12:42:37.478347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mpl.rc('font', size=14)\nmpl.rc('axes', titlesize=15)\nfigure, axes = plt.subplots(nrows=3, ncols=2) #3행 2열 figure 생성\nplt.tight_layout() # 그래프 사이 여백 확보\nfigure.set_size_inches(10,9) #전체 figure 사이즈 10x9인치로 설정\n\nsns.barplot(x='year', y='count', data=train, ax=axes[0, 0])\nsns.barplot(x='month', y='count', data= train, ax=axes[0, 1])\nsns.barplot(x='day', y='count', data= train, ax=axes[1, 0])\nsns.barplot(x='hour', y='count', data= train, ax=axes[1, 1])\nsns.barplot(x='minute', y='count', data= train, ax=axes[2, 0])\nsns.barplot(x='second', y='count', data= train, ax=axes[2, 1])\n\naxes[0, 0].set(title='Rental amounts by year')\naxes[0, 1].set(title='Rental amounts by month')\naxes[1, 0].set(title='Rental amounts by day')\naxes[1, 1].set(title='Rental amounts by hour')\naxes[2, 0].set(title='Rental amounts by minute')\naxes[2, 1].set(title='Rental amounts by seond')\n\naxes[1, 0].tick_params(axis='x', labelrotation=90)\naxes[1, 1].tick_params(axis='x', labelrotation=90)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:42:37.480608Z","iopub.execute_input":"2022-07-23T12:42:37.480906Z","iopub.status.idle":"2022-07-23T12:42:41.107715Z","shell.execute_reply.started":"2022-07-23T12:42:37.480866Z","shell.execute_reply":"2022-07-23T12:42:41.106804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#박스플롯은 범주형 데이터에 따른 수치형 데이터 정보를 나타냄, 막대그래프보다 많은 정보\n#계절,날씨,공휴일,근무일(범주형 데이터)에 따른 count(수치형 데이터)를 그려봄\n\nmpl.rc('font', size=14)\nmpl.rc('axes', titlesize=15)\nfigure, axes = plt.subplots(nrows=2, ncols=2) #2행 2열 figure 생성\nplt.tight_layout() # 그래프 사이 여백 확보\nfigure.set_size_inches(10,10) #전체 figure 사이즈 10x10인치로 설정\n\nsns.boxplot(x='season', y='count', data=train, ax=axes[0, 0])\nsns.boxplot(x='weather', y='count', data=train, ax=axes[0, 1])\nsns.boxplot(x='holiday', y='count', data=train, ax=axes[1, 0])\nsns.boxplot(x='workingday', y='count', data=train, ax=axes[1, 1])\n\naxes[0, 0].set(title='Box Plot on Count Across Season')\naxes[0, 1].set(title='Box Plot on Count Across Weather')\naxes[1, 0].set(title='Box Plot on Count Across Holiday')\naxes[1, 1].set(title='Box Plot on Count Across Workingday')\n\naxes[0,1].tick_params(axis='x', labelrotation=10)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:42:41.108839Z","iopub.execute_input":"2022-07-23T12:42:41.109057Z","iopub.status.idle":"2022-07-23T12:42:42.168187Z","shell.execute_reply.started":"2022-07-23T12:42:41.109032Z","shell.execute_reply":"2022-07-23T12:42:42.167172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#근무일, 공휴일, 요일, 계절, 날씨 (범주형 데이터)에 따른 시간대별 count(수치형 데이터)를 포인트플롯으로 표현\n#막대그래프와 동일한 평균과 신뢰구간을 나타내지만, 한 화면에 비교하기 편함 ,비교하고 싶은 피처를 hue 파라미터에 전달\n\nmpl.rc('font', size=11)\nfigure, axes = plt.subplots(nrows=5) # 5행 1열\nfigure.set_size_inches(12, 18)\n\nsns.pointplot(x='hour', y='count', data=train, hue='workingday', ax=axes[0])\nsns.pointplot(x='hour', y='count', data=train, hue='holiday', ax=axes[1])\nsns.pointplot(x='hour', y='count', data=train, hue='weekday', ax=axes[2])\nsns.pointplot(x='hour', y='count', data=train, hue='season', ax=axes[3])\nsns.pointplot(x='hour', y='count', data=train, hue='weather', ax=axes[4])","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:42:42.169550Z","iopub.execute_input":"2022-07-23T12:42:42.169885Z","iopub.status.idle":"2022-07-23T12:42:59.577093Z","shell.execute_reply.started":"2022-07-23T12:42:42.169853Z","shell.execute_reply":"2022-07-23T12:42:59.576487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#수치형 데이터인 온도, 체감 온도, 풍속, 습도별 대여수량을 '회귀선을 포함한 산점도 그래프'로 그려보기\n#수치형 데이터간 상관관계를 파악하는데 사용\nmpl.rc('font', size=15)\nfigure, axes = plt.subplots(nrows=2, ncols=2) # 2행 2열\nplt.tight_layout() # 그래프 간 공백 추가\nfigure.set_size_inches(7, 6)\n\n\n#scatter_kws는 점의 투명도 조절 (0 완전 투명, 0.2= 20퍼 투명, 1 불투명)\n\nsns.regplot(x='temp', y='count', data=train, ax=axes[0, 0],\n           scatter_kws={'alpha':0.2}, line_kws={'color':'blue'})\nsns.regplot(x='atemp', y='count', data=train, ax=axes[0, 1],\n           scatter_kws={'alpha':0.2}, line_kws={'color':'blue'})\nsns.regplot(x='windspeed', y='count', data=train, ax=axes[1, 0],\n           scatter_kws={'alpha':0.2}, line_kws={'color':'blue'})\nsns.regplot(x='humidity', y='count', data=train, ax=axes[1, 1],\n           scatter_kws={'alpha':0.2}, line_kws={'color':'blue'})\n\n#windspeed는 풍속이 0인 데이터가 많은데 관측치가 없거나 오류일 가능성 높음.\n#결측값이 많으면 상관관계 파악힘드므로 windspped 피처 삭제 예정","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:42:59.578225Z","iopub.execute_input":"2022-07-23T12:42:59.578937Z","iopub.status.idle":"2022-07-23T12:43:02.682488Z","shell.execute_reply.started":"2022-07-23T12:42:59.578904Z","shell.execute_reply":"2022-07-23T12:43:02.681775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# temp, atemp, humidity, windspeed, count(수치형 데이터) 끼리의 상관관계를 알아보기- heatmap\n# 수치형 데이터 간 상관관계 매트릭스\ntrain[['temp', 'atemp', 'humidity', 'windspeed', 'count']].corr()\n# 한눈에 들어오지 않으므로 히트맵 사용\ncorrMat=train[['temp', 'atemp', 'humidity', 'windspeed', 'count']].corr()\nfig, ax=plt.subplots()\nfig.set_size_inches(10, 10)\nsns.heatmap(corrMat, annot=True)\nax.set(title='Heatmap of Numerical Data')\n#상관계수의 절대값이 높을수록 상관관계 있다.\n#타깃값인 count와의 상관계수 비교해보기\n#windspeed는 0.1로 가장 작은 값, 상관관계가 적다. 게다가 결측값도 있으므로 제거 확","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:43:02.683636Z","iopub.execute_input":"2022-07-23T12:43:02.683960Z","iopub.status.idle":"2022-07-23T12:43:03.078005Z","shell.execute_reply.started":"2022-07-23T12:43:02.683932Z","shell.execute_reply":"2022-07-23T12:43:03.077441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> # **베이스라인 모델 훈련**\n\n1. 데이터 불러오기\n2. 피처 엔지니어링        \n       :   데이터 변환 하는 작업, 훈련 데이터와 테스트 데이터 공통으로 반영하기 위해서, 합쳤다가 끝나면 나눔\n 1. 훈련 데이터와 테스트 데이터 합치기\n 2. 피처 엔지니어링 ( 타입 변경, 삭제, 추가)\n 3. 데이터 나누기\n3. 평가지표 계산 함수 작성\n4. 모델 훈련\n5. 성능 검증\n6. 제출","metadata":{}},{"cell_type":"code","source":"# 데이터 불러오기\nimport numpy as np\nimport pandas as pd\n\ndata_path = '/kaggle/input/bike-sharing-demand/'\n\ntrain = pd.read_csv(data_path + 'train.csv')\ntest = pd.read_csv(data_path + 'test.csv')\nsubmission = pd.read_csv(data_path + 'sampleSubmission.csv')\n\n# 이상치 제거\n# weather가 4인 데이터(폭우, 폭설 데이터)는 이상치였기 때문에 삭제\ntrain = train[train['weather'] != 4]\n\n# 데이터 합치기\n# 판다스의 concat() 함수를 이용하면 축을 따라 DataFrame을 이어붙일 수 있음\n\nall_data=pd.concat([train, test], ignore_index=True) #ignore_index=True는 제거한 인덱스 무시하고 이어붙이기\nall_data\n\n# 파생 피처(변수) 추가\n# day 피처는 사용 X (19일 이후 정보 없음), minute, second 피처 사용 X(0으로 일정)\nfrom datetime import datetime\nall_data['date'] = all_data['datetime'].apply(lambda x: x.split()[0])\nall_data['year'] = all_data['datetime'].apply(lambda x: x.split()[0].split('-')[0])\nall_data['month'] = all_data['datetime'].apply(lambda x: x.split()[0].split('-')[1])\nall_data['hour'] = all_data['datetime'].apply(lambda x: x.split()[1].split(':')[0])\nall_data['weekday'] = all_data['date'].apply(lambda dataString :\n                                                 datetime.strptime(dataString,\"%Y-%m-%d\").weekday())\n\n# 필요 없는 피처 제거\n# casual, registered는 테스트에 없는 피처- > 제거\n# datetime와 date는 인덱스와 year month day에 담겨있는 중복 정보 -> 제거\n# windspeed 타깃값 count와 상관관계 적음 -> 제거\ndrop_features = ['casual', 'registered', 'datetime', 'date', 'windspeed', 'month']\nall_data= all_data.drop(drop_features, axis=1)\n\nall_data\n\n# 데이터 나누기\nX_train= all_data[~pd.isnull(all_data['count'])] #책 오타 있네요 , ~는 부정의 듯\nX_test= all_data[pd.isnull(all_data['count'])]\n\n# 타깃값 count 제거\nX_train=X_train.drop(['count'], axis=1)\nX_test=X_test.drop(['count'], axis=1)\n\ny= train['count'] # 타깃값\n\nX_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:43:03.079145Z","iopub.execute_input":"2022-07-23T12:43:03.079518Z","iopub.status.idle":"2022-07-23T12:43:03.432782Z","shell.execute_reply.started":"2022-07-23T12:43:03.079488Z","shell.execute_reply":"2022-07-23T12:43:03.431797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#평가지표 계산 함수 작성\nimport numpy as np\n\ndef rmsle(y_true, y_pred, convertExp=True):\n    # 지수변환\n    if convertExp:\n        y_true=np.exp(y_true)\n        y_pred=np.exp(y_pred)\n        \n    # 로그변환 후 결측값을 0으로 변환\n    log_true=np.nan_to_num(np.log(y_true+1))\n    log_pred=np.nan_to_num(np.log(y_pred+1))\n    \n    #RMSLE 계산\n    output =np.sqrt(np.mean((log_true - log_pred)**2))\n    return output","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:43:03.434149Z","iopub.execute_input":"2022-07-23T12:43:03.435111Z","iopub.status.idle":"2022-07-23T12:43:03.443023Z","shell.execute_reply.started":"2022-07-23T12:43:03.435063Z","shell.execute_reply":"2022-07-23T12:43:03.442171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#모델 훈련\nfrom sklearn.linear_model import LinearRegression\n\nlinear_reg_model = LinearRegression() #선형 회귀 모델\n\nlog_y=np.log(y) #타깃값 로그변환\nlinear_reg_model.fit(X_train, log_y) #모델 훈련\n\n#모델 성능 검증\npreds = linear_reg_model.predict(X_train) #타깃값 예측\n\nprint(f'선형 회귀의 RMSLE 값 : {rmsle(log_y, preds, True):.4f}')\n\n\n#예측 및 결과 제출\nlinearreg_preds= linear_reg_model.predict(X_test)\n\nsubmission['count']=np.exp(linearreg_preds) #지수변환\nsubmission.to_csv('submission.csv', index=False) #파일로 저장","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:43:03.444143Z","iopub.execute_input":"2022-07-23T12:43:03.444476Z","iopub.status.idle":"2022-07-23T12:43:03.581942Z","shell.execute_reply.started":"2022-07-23T12:43:03.444448Z","shell.execute_reply":"2022-07-23T12:43:03.580963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> # **성능 개선**\n1. 데이터 불러오기\n2. 피처 엔지니어링        \n       :   데이터 변환 하는 작업, 훈련 데이터와 테스트 데이터 공통으로 반영하기 위해서, 합쳤다가 끝나면 나눔\n 1. 훈련 데이터와 테스트 데이터 합치기\n 2. 피처 엔지니어링 ( 타입 변경, 삭제, 추가)\n 3. 데이터 나누기\n3. 평가지표 계산 함수 작성\n4. 하이퍼파라미터 최적화(모델 훈련)  \n    1.     모델 생성(릿지 회귀)  \n    2.그리드서치 객체 생성\n    3.훈련(그리드 서치)\n5. 성능 검증\n6. 제출","metadata":{}},{"cell_type":"markdown","source":"# **릿지 회귀 모델**","metadata":{}},{"cell_type":"code","source":"# 하이퍼파라미터 최적화(모델 훈련)\n# A. 모델 생성\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn import metrics\n\nridge_model = Ridge()\n# B. 그리드서치 객체 생성(최적 파라미터 찾기 위해)\n# 릿지 모델의 하이퍼파라미터는 alpha로 값이 클수록 규제 강도가 세짐\n\n#하이퍼파라미터 값 목록\nridge_params ={'max_iter':[3000], 'alpha':[0.1,1,2,3,4,10,30,100,200,300,400,800,900,1000]}\n\n# 교차 검증용 평가 함수\nrmsle_scorer=metrics.make_scorer(rmsle, greater_is_better=False)\n\n# 그리드(with 릿지) 객체 생성\ngridsearch_ridge_model=GridSearchCV(estimator=ridge_model,   #릿지 모델\n                                    param_grid=ridge_params, #값 목록\n                                    scoring=rmsle_scorer,    #평가지표\n                                    cv=5)                    #교차 검증 분할 수\n\nlog_y= np.log(y) #타깃값 로그변환\ngridsearch_ridge_model.fit(X_train, log_y) #훈련 (그리드서치)\nprint('최적 하이퍼파라미터 : ', gridsearch_ridge_model.best_params_)\n\n#성능 검증\n#예측\npreds=gridsearch_ridge_model.best_estimator_.predict(X_train)\n#평가\nprint(f'릿지 회귀 RMSLE 값 : {rmsle(log_y, preds, True):.4f}')\n\n#결과가 크게 달라지지 않음..","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:45:32.706063Z","iopub.execute_input":"2022-07-23T12:45:32.706777Z","iopub.status.idle":"2022-07-23T12:45:35.285791Z","shell.execute_reply.started":"2022-07-23T12:45:32.706736Z","shell.execute_reply":"2022-07-23T12:45:35.284767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **라쏘 회귀 모델**","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import Lasso\n\n#모델 생성\nlasso_model = Lasso()\n#하이퍼파라미터 값 목록\nlasso_alpha=1/np.array([0.1,1,2,3,4,10,30,100,200,300,400,800,900,1000])\nlasso_params={'max_iter':[3000], 'alpha':lasso_alpha}\n#그리드서치(with 라쏘) 객체 생성\ngridsearch_lasso_model=GridSearchCV(estimator=lasso_model,\n                                    param_grid=lasso_params,\n                                    scoring=rmsle_scorer,\n                                    cv=5)\n\n#그리드서치 수행\nlog_y=np.log(y)\ngridsearch_lasso_model.fit(X_train, log_y)\n\nprint('최적 하이퍼파라미터 :', gridsearch_lasso_model.best_params_)\n\n#성능 검증\n#예측\npreds=gridsearch_lasso_model.best_estimator_.predict(X_train)\n#평가\nprint(f'라쏘 회귀 RMSLE 값 : {rmsle(log_y, preds, True):.4f}')\n\n# 값 변화 X...","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:01:09.610418Z","iopub.execute_input":"2022-07-23T13:01:09.610702Z","iopub.status.idle":"2022-07-23T13:01:13.337119Z","shell.execute_reply.started":"2022-07-23T13:01:09.610674Z","shell.execute_reply":"2022-07-23T13:01:13.334057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **랜덤 포레스트 회귀 모델**\n훈련 데이터를 랜덤하게 샘플링한 모델 n개를 각각 훈련하여 결과를 평균하는 방법","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\n\n#모델 생성\nrandomforest_model=RandomForestRegressor()\n#그리드서치 객체 생성\nrf_params={'random_state':[42], 'n_estimators':[100, 120, 140]}  #랜덤 포레스트 회귀 모델의 파라미터는 random_state(랜덤 시드값..)와 n_estimators(결정 트리 개수)\ngridsearch_random_forest_model = GridSearchCV(estimator=randomforest_model,\n                                              param_grid=rf_params,\n                                              scoring=rmsle_scorer,\n                                              cv=5)\n#그리드서치 수행\nlog_y=np.log(y)\ngridsearch_random_forest_model.fit(X_train, log_y)\nprint('최적 하이퍼파라미터 : ', gridsearch_random_forest_model.best_params_)\n\n#예측\npreds=gridsearch_random_forest_model.best_estimator_.predict(X_train)\n\n#평가\nprint(f'랜덤 포레스트 회귀 RMSLE 값 : {rmsle(log_y, preds, True):4f}')\n\n# 시간이 걸리지만, 가장 결과가 좋게 나왔음!","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:15:30.365824Z","iopub.execute_input":"2022-07-23T13:15:30.366200Z","iopub.status.idle":"2022-07-23T13:16:26.343634Z","shell.execute_reply.started":"2022-07-23T13:15:30.366163Z","shell.execute_reply":"2022-07-23T13:16:26.342692Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nrandomforest_preds=gridsearch_random_forest_model.best_estimator_.predict(X_test)\n\nsubmission['count']=np.exp(randomforest_preds) #지수변환\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:19:24.434763Z","iopub.execute_input":"2022-07-23T13:19:24.435081Z","iopub.status.idle":"2022-07-23T13:19:24.657415Z","shell.execute_reply.started":"2022-07-23T13:19:24.435049Z","shell.execute_reply":"2022-07-23T13:19:24.656504Z"},"trusted":true},"execution_count":null,"outputs":[]}]}