{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-14T03:38:27.480034Z","iopub.execute_input":"2022-07-14T03:38:27.480927Z","iopub.status.idle":"2022-07-14T03:38:27.494801Z","shell.execute_reply.started":"2022-07-14T03:38:27.480826Z","shell.execute_reply":"2022-07-14T03:38:27.493668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Bike Sharing Demand 진행방향\n- 1) 훈련, 테스트 데이터셋의 형태 및 컬럼의 속성 데이터 값 파악\n- 2) 데이터 전처리 및 시각화\n- 3) 회귀모델 적용\n- 4) 결론 도출","metadata":{}},{"cell_type":"markdown","source":"## Step 01. 필수 라이브러리 불러오기\n- 주요 라이브러리 버전을 확인한다.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib as mpl\nimport calendar\nfrom datetime import datetime\nimport os\nprint(os.listdir(\"../input\"))\nimport matplotlib.pyplot as plt\n# 경고문 무시\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n# 버전 확인\nprint(\"pandas version : \", pd.__version__)\nprint(\"numpy version : \", np.__version__)\nprint(\"seaborn version: \", sns.__version__)\nprint(\"matplotlib version : \", mpl.__version__)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:27.501802Z","iopub.execute_input":"2022-07-14T03:38:27.502224Z","iopub.status.idle":"2022-07-14T03:38:27.996797Z","shell.execute_reply.started":"2022-07-14T03:38:27.502182Z","shell.execute_reply":"2022-07-14T03:38:27.995500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 02. 데이터 불러오기\n- 데이터를 불러온다. ","metadata":{}},{"cell_type":"code","source":"DATA_PATH = '/kaggle/input/bike-sharing-demand/'\ntrain = pd.read_csv(DATA_PATH + \"train.csv\")\ntest = pd.read_csv(DATA_PATH + \"test.csv\")\nsubmission = pd.read_csv(DATA_PATH + \"sampleSubmission.csv\")\n\nprint(\"데이터를 불러왔습니다.\")","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:27.998341Z","iopub.execute_input":"2022-07-14T03:38:27.998661Z","iopub.status.idle":"2022-07-14T03:38:28.048479Z","shell.execute_reply.started":"2022-07-14T03:38:27.998626Z","shell.execute_reply":"2022-07-14T03:38:28.047284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## step 03. 데이터 확인하기\n- 데이터에 관해 궁금한 점이 있으면 캐들 대회처의 Overview(개요)를 둘러보면 된다.","metadata":{}},{"cell_type":"markdown","source":"## 데이터 셋 내에 있는 컬럼 속성들에 대한 설명\n- datetime - hourly date + timestamp  \n- season -  1 = spring, 2 = summer, 3 = fall, 4 = winter \n- holiday - whether the day is considered a holiday\n- workingday - whether the day is neither a weekend nor holiday\n- weather - 1: Clear, Few clouds, Partly cloudy, Partly cloudy \n          + 2: Mist + Cloudy, Mist + Broken clouds, Mist + Few clouds, Mist \n          + 3: Light Snow, Light Rain + Thunderstorm + Scattered clouds, Light Rain + Scattered clouds \n          + 4: Heavy Rain + Ice Pallets + Thunderstorm + Mist, Snow + Fog \n- temp - temperature in Celsius\n- atemp - \"feels like\" temperature in Celsius\n- humidity - relative humidity\n- windspeed - wind speed\n- casual - number of non-registered user rentals initiated\n- registered - number of registered user rentals initiated\n- count - number of total rentals","metadata":{}},{"cell_type":"code","source":"#훈련데이터 셋의 개괄적인 모형 파악\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:28.051162Z","iopub.execute_input":"2022-07-14T03:38:28.051536Z","iopub.status.idle":"2022-07-14T03:38:28.070952Z","shell.execute_reply.started":"2022-07-14T03:38:28.051506Z","shell.execute_reply":"2022-07-14T03:38:28.069768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#테스트 데이터 셋의 개괄적인 형태 출력\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:28.072780Z","iopub.execute_input":"2022-07-14T03:38:28.073099Z","iopub.status.idle":"2022-07-14T03:38:28.087581Z","shell.execute_reply.started":"2022-07-14T03:38:28.073070Z","shell.execute_reply":"2022-07-14T03:38:28.086338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 04. 데이터 전처리 및 시각화","metadata":{}},{"cell_type":"code","source":"#datetime속성을 분리하여 추출속성으로 활용하기 위해 split함수를 사용하여 년-월-일 과 시간을 분리한다.\ntrain['tempDate'] = train.datetime.apply(lambda x:x.split())\n\nprint('실행 완료')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:28.088718Z","iopub.execute_input":"2022-07-14T03:38:28.089192Z","iopub.status.idle":"2022-07-14T03:38:28.107309Z","shell.execute_reply.started":"2022-07-14T03:38:28.089154Z","shell.execute_reply":"2022-07-14T03:38:28.105977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#분리한 tempDate를 가지고 년-월-일을 이용하여 year,month,day 그리고 weekday column을 추출한다.\n# split() 내장함수 설명: https://wikidocs.net/13 [문자형 자료형_ 문자열 나누기] <=> join() [문자형 자료형_ 문자열 삽입]\ntrain['year'] = train.tempDate.apply(lambda x:x[0].split('-')[0])\ntrain['month'] = train.tempDate.apply(lambda x:x[0].split('-')[1])\ntrain['day'] = train.tempDate.apply(lambda x:x[0].split('-')[2])\n\n\n#weekday는 calendar패키지와 datetime패키지를 활용한다.\n#calendar.day_name 사용법 : https://stackoverflow.com/questions/36341484/get-day-name-from-weekday-int\n#datetime.strptime 문서: https://docs.python.org/3/library/datetime.html#strftime-strptime-behavior\n#파이썬에서 날짜와 시간 다루기: https://datascienceschool.net/view-notebook/465066ac92ef4da3b0aba32f76d9750a/ \ntrain['weekday'] = train.tempDate.apply(lambda x:calendar.day_name[datetime.strptime(x[0], \"%Y-%m-%d\").weekday()])\n\ntrain['hour'] = train.tempDate.apply(lambda x:x[1].split(':')[0])\n\nprint('실행 완료')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:28.108476Z","iopub.execute_input":"2022-07-14T03:38:28.109013Z","iopub.status.idle":"2022-07-14T03:38:28.271216Z","shell.execute_reply.started":"2022-07-14T03:38:28.108972Z","shell.execute_reply":"2022-07-14T03:38:28.269971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#분리를 통해 추출된 속성은 문자열 속성을 가지고 있음 따라서 숫자형 데이터로 변환해 줄 필요가 있음.\n#pandas.to_numeric(): https://pandas.pydata.org/pandas-docs/stable/generated/pandas.to_numeric.html\n\ntrain['year'] = pd.to_numeric(train.year,errors='coerce')\ntrain['month'] = pd.to_numeric(train.month,errors='coerce')\ntrain['day'] = pd.to_numeric(train.day,errors='coerce')\ntrain['hour'] = pd.to_numeric(train.hour,errors='coerce')\n\nprint('실행 완료')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:28.272829Z","iopub.execute_input":"2022-07-14T03:38:28.273937Z","iopub.status.idle":"2022-07-14T03:38:28.307918Z","shell.execute_reply.started":"2022-07-14T03:38:28.273890Z","shell.execute_reply":"2022-07-14T03:38:28.307137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#year,month,day,hour가 숫자형으로 변환되었음을 알 수 있음.\ntrain.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:28.309014Z","iopub.execute_input":"2022-07-14T03:38:28.309512Z","iopub.status.idle":"2022-07-14T03:38:28.326441Z","shell.execute_reply.started":"2022-07-14T03:38:28.309482Z","shell.execute_reply":"2022-07-14T03:38:28.325030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#필요를 다한 tempDate column을 drop함\ntrain = train.drop('tempDate', axis=1)\n\nprint('실행 완료')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:28.327828Z","iopub.execute_input":"2022-07-14T03:38:28.328146Z","iopub.status.idle":"2022-07-14T03:38:28.338462Z","shell.execute_reply.started":"2022-07-14T03:38:28.328118Z","shell.execute_reply":"2022-07-14T03:38:28.337296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#각각의 속성과 예측의 결과값으로 쓰이는 count값과의 관계 파악\n\n# year와 count\nfig = plt.figure(figsize=[12,10])\nax1 = fig.add_subplot(2,2,1)\nax1 = sns.barplot(x='year', y='count', data=train.groupby('year')['count'].mean().reset_index())\n\n# month와 count\nax2 = fig.add_subplot(2,2,2)\nax2 = sns.barplot(x='month', y='count', data=train.groupby('month')['count'].mean().reset_index())\n\n# day와 count\nax3 = fig.add_subplot(2,2,3)\nax3 = sns.barplot(x='day', y='count', data=train.groupby('day')['count'].mean().reset_index())\n\n\n# hour와 count\nax4 = fig.add_subplot(2,2,4)\nax4 = sns.barplot(x='hour', y='count', data=train.groupby('hour')['count'].mean().reset_index())","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:28.339788Z","iopub.execute_input":"2022-07-14T03:38:28.340577Z","iopub.status.idle":"2022-07-14T03:38:29.382405Z","shell.execute_reply.started":"2022-07-14T03:38:28.340540Z","shell.execute_reply":"2022-07-14T03:38:29.379713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# season과 count\nfig = plt.figure(figsize=[12,10])\nax1 = fig.add_subplot(2,2,1)\nax1 = sns.barplot(x='season', y='count', data = train.groupby('season')['count'].mean().reset_index())\n\n\n# holiday 여부와 count\nax2 = fig.add_subplot(2,2,2)\nax2 = sns.barplot(x='holiday', y='count', data=train.groupby('holiday')['count'].mean().reset_index())\n\n\n# workingday 여부와 count\nax3 = fig.add_subplot(2,2,3)\nax3 = sns.barplot(x='workingday', y='count', data=train.groupby('workingday')['count'].mean().reset_index())\n\n\n# weather와 count\nax4 = fig.add_subplot(2,2,4)\nax4 = sns.barplot(x='weather', y='count', data=train.groupby('weather')['count'].mean().reset_index())","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:29.386542Z","iopub.execute_input":"2022-07-14T03:38:29.387650Z","iopub.status.idle":"2022-07-14T03:38:29.901872Z","shell.execute_reply.started":"2022-07-14T03:38:29.387598Z","shell.execute_reply":"2022-07-14T03:38:29.900427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n해당 부분은 필자가 스스로 데이터를 보고 이상함을 느껴 전처리함.\n왜냐하면, 처음 import한 데이터 셋에서 head()를 하였을 때 1월1일의 season column은 1 즉 봄을 가르키는데,\n직접 3월에 washington을 직접 가본 결과 1월은 확실히 겨울이다.\n따라서 아래의 badToRight를 이용하여 season column을 수정하고자 했음.\n이 데이터 때문에 참조했던 커널과는 다른 정확도를 나타낼 수 있음.\n\"\"\"\n\n\ndef badToRight(month):\n    if month in [12,1,2]:\n        return 4\n    elif month in [3,4,5]:\n        return 1\n    elif month in [6,7,8]:\n        return 2\n    elif month in [9,10,11]:\n        return 3\n    \n#apply() 내장함수는 split(),map(),join(),filter()등 과 함꼐 필수적으로 숙지해야 할 함수이다.\ntrain['season'] = train.month.apply(badToRight)\n\nprint('실행 완료')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:29.908967Z","iopub.execute_input":"2022-07-14T03:38:29.909555Z","iopub.status.idle":"2022-07-14T03:38:29.927950Z","shell.execute_reply.started":"2022-07-14T03:38:29.909511Z","shell.execute_reply":"2022-07-14T03:38:29.925827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#위의 시각화와 같이 하나의 컬럼과 결과 값을 비교해보자\n\n# season과 count\nfig = plt.figure(figsize=[12,10])\nax1 = fig.add_subplot(2,2,1)\nax1 = sns.barplot(x='season', y='count', data = train.groupby('season')['count'].mean().reset_index())\n\n\n# holiday 여부와 count\nax2 = fig.add_subplot(2,2,2)\nax2 = sns.barplot(x='holiday', y='count', data=train.groupby('holiday')['count'].mean().reset_index())\n\n\n# workingday 여부와 count\nax3 = fig.add_subplot(2,2,3)\nax3 = sns.barplot(x='workingday', y='count', data=train.groupby('workingday')['count'].mean().reset_index())\n\n\n# weather와 count\nax4 = fig.add_subplot(2,2,4)\nax4 = sns.barplot(x='weather', y='count', data=train.groupby('weather')['count'].mean().reset_index())","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:29.929310Z","iopub.execute_input":"2022-07-14T03:38:29.929747Z","iopub.status.idle":"2022-07-14T03:38:30.461039Z","shell.execute_reply.started":"2022-07-14T03:38:29.929700Z","shell.execute_reply":"2022-07-14T03:38:30.459738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#그리고 남은 분포를 통해 표현하였을 때 좋은 컬럼들을 count와 비교해보자\n\n# temp와 count\nfig = plt.figure(figsize=[12,10])\nax1 = fig.add_subplot(2,2,1)\nax1 = sns.distplot(train.temp,bins=range(train.temp.min().astype('int'),train.temp.max().astype('int')+1))\n\n# atemp와 count\nax2 = fig.add_subplot(2,2,2)\nax2 = sns.distplot(train.atemp,bins=range(train.atemp.min().astype('int'),train.atemp.max().astype('int')+1))\n\n# humidity와 count\nax3 = fig.add_subplot(2,2,3)\nax3 = sns.distplot(train.humidity,bins=range(train.humidity.min().astype('int'),train.humidity.max().astype('int')+1))\n\n# windspeed와 count\nax4 = fig.add_subplot(2,2,4)\nax4 = sns.distplot(train.windspeed,bins=range(train.windspeed.min().astype('int'),train.windspeed.max().astype('int')+1))","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:30.462765Z","iopub.execute_input":"2022-07-14T03:38:30.463279Z","iopub.status.idle":"2022-07-14T03:38:32.178248Z","shell.execute_reply.started":"2022-07-14T03:38:30.463220Z","shell.execute_reply":"2022-07-14T03:38:32.176902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#각각의 컬럼들 간의 상관계수를 heatmap을 통해 시각화\n\nfig = plt.figure(figsize =[20,20])\nax = sns.heatmap(train.corr(), annot=True, square=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:32.180362Z","iopub.execute_input":"2022-07-14T03:38:32.181213Z","iopub.status.idle":"2022-07-14T03:38:33.453324Z","shell.execute_reply.started":"2022-07-14T03:38:32.181156Z","shell.execute_reply":"2022-07-14T03:38:33.451052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.weekday","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:33.455650Z","iopub.execute_input":"2022-07-14T03:38:33.457630Z","iopub.status.idle":"2022-07-14T03:38:33.476808Z","shell.execute_reply.started":"2022-07-14T03:38:33.457563Z","shell.execute_reply":"2022-07-14T03:38:33.475589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#heatmap 상관관계를 참조하여 이전의 시각화와는 달리 두 개의 서로다른 컬럼이 적용된 count를 시각화해보자\n\n# hour과 season에 따른 count\nfig = plt.figure(figsize=[12,10])\nax1 = fig.add_subplot(2,2,1)\nax1 = sns.pointplot(x='hour', y='count', hue='season', data=train.groupby(['season', 'hour'])['count'].mean().reset_index())\n\n# hour과 holiday 여부에 따른 count\nax2 = fig.add_subplot(2,2,2)\nax2 = sns.pointplot(x='hour', y='count', hue='holiday', data=train.groupby(['holiday', 'hour'])['count'].mean().reset_index())\n\n# hour과 weekday(요일) 여부에 따른 count\nax3 = fig.add_subplot(2,2,3)\nax3 = sns.pointplot(x='hour', y='count', hue='weekday', hue_order=['Sunday', 'Monday', 'Tuesday', 'Wednesday', 'Thursday', 'Friday', 'Saturday'],\n                    data=train.groupby(['weekday', 'hour'])['count'].mean().reset_index())\n\n# hour과 weather에 따른 count\nax4 = fig.add_subplot(2,2,4)\nax4 = sns.pointplot(x='hour', y='count', hue='weather', data=train.groupby(['weather', 'hour'])['count'].mean().reset_index())","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:33.478204Z","iopub.execute_input":"2022-07-14T03:38:33.479564Z","iopub.status.idle":"2022-07-14T03:38:36.354949Z","shell.execute_reply.started":"2022-07-14T03:38:33.479504Z","shell.execute_reply":"2022-07-14T03:38:36.353578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#마지막 시각화에 이상치가 있는 것같아서 확인\n\ntrain[train.weather==4]","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:36.357793Z","iopub.execute_input":"2022-07-14T03:38:36.358346Z","iopub.status.idle":"2022-07-14T03:38:36.388253Z","shell.execute_reply.started":"2022-07-14T03:38:36.358290Z","shell.execute_reply":"2022-07-14T03:38:36.386685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# month와 날씨에 따른 count \nfig = plt.figure(figsize=[12,10])\nax1 = fig.add_subplot(2,1,1)\nax1 = sns.pointplot(x='month', y='count', hue='weather', data=train.groupby(['weather','month'])['count'].mean().reset_index())\n\n# 월별 count\nax2 = fig.add_subplot(2,1,2)\nax2 = sns.barplot(x='month', y='count', data=train.groupby(['month'])['count'].mean().reset_index())\n","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:36.391156Z","iopub.execute_input":"2022-07-14T03:38:36.392452Z","iopub.status.idle":"2022-07-14T03:38:36.951359Z","shell.execute_reply.started":"2022-07-14T03:38:36.392379Z","shell.execute_reply":"2022-07-14T03:38:36.949953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nWindspeed 분포를 표현한 그래프에서 Windspeed가 0인 값들이 많았는데,\n이는 실제로 0이었던지 or 값을 제대로 측정하지 못해서 0인지 두 개의 경우가 있다.\n하지만 후자의 생각을 가지고 우리의 데이터를 활용하여 windspeed값을 부여해보자\n\"\"\"\n\n#머신러닝 모델에 훈련시킬 때는 문자열 값은 불가능하기 때문에 문자열을 카테고리화 하고 각각에 해당하는 값을 숫자로 변환해준다\ntrain['weekday'] = train.weekday.astype('category')\n\nprint('실행 완료')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:36.953361Z","iopub.execute_input":"2022-07-14T03:38:36.953758Z","iopub.status.idle":"2022-07-14T03:38:36.964957Z","shell.execute_reply.started":"2022-07-14T03:38:36.953723Z","shell.execute_reply":"2022-07-14T03:38:36.963570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train['weekday'].cat.categories)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:36.966737Z","iopub.execute_input":"2022-07-14T03:38:36.967159Z","iopub.status.idle":"2022-07-14T03:38:36.978455Z","shell.execute_reply.started":"2022-07-14T03:38:36.967123Z","shell.execute_reply":"2022-07-14T03:38:36.976818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#0:Sunday --> 6:Saturday\ntrain.weekday.cat.categories = [ '5', '1', '6', '0', '4', '2', '3']\n\nprint('실행 완료')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:36.980361Z","iopub.execute_input":"2022-07-14T03:38:36.980802Z","iopub.status.idle":"2022-07-14T03:38:36.991689Z","shell.execute_reply.started":"2022-07-14T03:38:36.980764Z","shell.execute_reply":"2022-07-14T03:38:36.990521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nRandomForest를 활용하여 Windspeed값을 부여해보자\n하나의 데이터를 Windspeed가 0인 그리고 0이 아닌 데이터프레임으로 분리하고\n학습시킬 0이 아닌 데이터 프레임에서는 Windspeed만 담긴 Series와 이외의 학습시킬 column들의 데이터프레임으로 분리한다\n학습 시킨 후에 Windspeed가 0인 데이터 프레임에서 학습시킨 컬럼과 같게 추출하여 결과 값을 부여받은 후,\nWindspeed가 0인 데이터프레임에 Windspeed값을 부여한다.\n\"\"\"\nfrom sklearn.ensemble import RandomForestRegressor\n\n#Windspeed가 0인 데이터프레임\nwindspeed_0 = train[train.windspeed ==0]\n#Windspeed가 0이 아닌 데이터프레임\nwindspeed_Not0 = train[train.windspeed != 0]\n\n#Windspeed가 0인 데이터 프레임에 투입을 원치 않는 컬럼을 배제\nwindspeed_0_df = windspeed_0.drop(['windspeed','casual','registered','count','datetime'],axis=1)\n\n#Windspeed가 0이 아닌 데이터 프레임은 위와 동일한 데이터프레임을 형성하고 학습시킬 Windspeed Series를 그대로 둠\nwindspeed_Not0_df = windspeed_Not0.drop(['windspeed','casual','registered','count','datetime'],axis=1)\nwindspeed_Not0_series = windspeed_Not0['windspeed'] \n\n#모델에 0이 아닌 데이터프레임과 결과값을 학습\nrf = RandomForestRegressor()\nrf.fit(windspeed_Not0_df,windspeed_Not0_series)\n#학습된 모델에 Windspeed가 0인 데이터프레임의 Windspeed를 도출\npredicted_windspeed_0 = rf.predict(windspeed_0_df)\n#도출된 값을 원래의 데이터프레임에 삽입\nwindspeed_0['windspeed'] = predicted_windspeed_0\n\nprint('실행 완료')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:36.993202Z","iopub.execute_input":"2022-07-14T03:38:36.993877Z","iopub.status.idle":"2022-07-14T03:38:40.660383Z","shell.execute_reply.started":"2022-07-14T03:38:36.993837Z","shell.execute_reply":"2022-07-14T03:38:40.659057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#나눈 데이터 프레임을 원래의 형태로 복원\ntrain = pd.concat([windspeed_0,windspeed_Not0], axis=0)\n\nprint('실행 완료')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:40.661956Z","iopub.execute_input":"2022-07-14T03:38:40.662334Z","iopub.status.idle":"2022-07-14T03:38:40.674174Z","shell.execute_reply.started":"2022-07-14T03:38:40.662292Z","shell.execute_reply":"2022-07-14T03:38:40.672795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#시간별 정렬을 위해 string type의 datetime을 datetime으로 변환\ntrain.datetime = pd.to_datetime(train.datetime,errors='coerce')\n\nprint('실행 완료')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:40.675969Z","iopub.execute_input":"2022-07-14T03:38:40.677340Z","iopub.status.idle":"2022-07-14T03:38:40.689845Z","shell.execute_reply.started":"2022-07-14T03:38:40.677250Z","shell.execute_reply":"2022-07-14T03:38:40.687551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#합쳐진 데이터를 datetime순으로 정렬\ntrain = train.sort_values(by=['datetime'])\n\nprint('실행 완료')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:40.691516Z","iopub.execute_input":"2022-07-14T03:38:40.693035Z","iopub.status.idle":"2022-07-14T03:38:40.701499Z","shell.execute_reply.started":"2022-07-14T03:38:40.692991Z","shell.execute_reply":"2022-07-14T03:38:40.700200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#windspeed를 수정한 후 다시 상관계수를 분석\n#우리의 기대와는 달리 windspeed와 count의 상관관계는 0.1에서 0.11로 간소한 차이만 보임.\nfig = plt.figure(figsize=[20,20])\nax = sns.heatmap(train.corr(), annot=True, square=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:40.702950Z","iopub.execute_input":"2022-07-14T03:38:40.703786Z","iopub.status.idle":"2022-07-14T03:38:41.930871Z","shell.execute_reply.started":"2022-07-14T03:38:40.703735Z","shell.execute_reply":"2022-07-14T03:38:41.929603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=[5,5])\nsns.distplot(train['windspeed'],bins=np.linspace(train['windspeed'].min(),train['windspeed'].max(),10))\nplt.suptitle(\"Filled by Random Forest Regressor\")\nprint(\"Min value of windspeed is {}\".format(train['windspeed'].min()))","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:41.932666Z","iopub.execute_input":"2022-07-14T03:38:41.933311Z","iopub.status.idle":"2022-07-14T03:38:42.208799Z","shell.execute_reply.started":"2022-07-14T03:38:41.933227Z","shell.execute_reply":"2022-07-14T03:38:42.207638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"이제 모든 동일한 전처리 과정을 test셋과 한꺼번에 진행\"\"\"\ntrain = pd.read_csv(DATA_PATH + \"train.csv\")\ntest = pd.read_csv(DATA_PATH + \"test.csv\")\n\nprint('실행 완료')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:42.210374Z","iopub.execute_input":"2022-07-14T03:38:42.211510Z","iopub.status.idle":"2022-07-14T03:38:42.247847Z","shell.execute_reply.started":"2022-07-14T03:38:42.211467Z","shell.execute_reply":"2022-07-14T03:38:42.246419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combine = pd.concat([train,test], axis=0)\n\nprint('실행 완료')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:42.249726Z","iopub.execute_input":"2022-07-14T03:38:42.250391Z","iopub.status.idle":"2022-07-14T03:38:42.262816Z","shell.execute_reply.started":"2022-07-14T03:38:42.250340Z","shell.execute_reply":"2022-07-14T03:38:42.261400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combine.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:42.264853Z","iopub.execute_input":"2022-07-14T03:38:42.265804Z","iopub.status.idle":"2022-07-14T03:38:42.283212Z","shell.execute_reply.started":"2022-07-14T03:38:42.265750Z","shell.execute_reply":"2022-07-14T03:38:42.281873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- 시간 데이터 전처리","metadata":{}},{"cell_type":"code","source":"combine['tempDate'] = combine.datetime.apply(lambda x:x.split())\ncombine['weekday'] = combine.tempDate.apply(lambda x: calendar.day_name[datetime.strptime(x[0],\"%Y-%m-%d\").weekday()])\ncombine['year'] = combine.tempDate.apply(lambda x: x[0].split('-')[0])\ncombine['month'] = combine.tempDate.apply(lambda x: x[0].split('-')[1])\ncombine['day'] = combine.tempDate.apply(lambda x: x[0].split('-')[2])\ncombine['hour'] = combine.tempDate.apply(lambda x: x[1].split(':')[0])","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:42.286440Z","iopub.execute_input":"2022-07-14T03:38:42.287643Z","iopub.status.idle":"2022-07-14T03:38:42.742936Z","shell.execute_reply.started":"2022-07-14T03:38:42.287580Z","shell.execute_reply":"2022-07-14T03:38:42.741717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combine['year'] = pd.to_numeric(combine.year,errors='coerce')\ncombine['month'] = pd.to_numeric(combine.month,errors='coerce')\ncombine['day'] = pd.to_numeric(combine.day,errors='coerce')\ncombine['hour'] = pd.to_numeric(combine.hour,errors='coerce')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:42.744669Z","iopub.execute_input":"2022-07-14T03:38:42.745048Z","iopub.status.idle":"2022-07-14T03:38:42.788956Z","shell.execute_reply.started":"2022-07-14T03:38:42.745007Z","shell.execute_reply":"2022-07-14T03:38:42.787563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combine.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:42.790587Z","iopub.execute_input":"2022-07-14T03:38:42.790953Z","iopub.status.idle":"2022-07-14T03:38:42.812116Z","shell.execute_reply.started":"2022-07-14T03:38:42.790923Z","shell.execute_reply":"2022-07-14T03:38:42.811141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- 계절 데이터 전처리 (계절에 해당되는 월 교정)","metadata":{}},{"cell_type":"code","source":"combine['season'] = combine.month.apply(badToRight)\n\nprint('실행 완료')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:42.815111Z","iopub.execute_input":"2022-07-14T03:38:42.815647Z","iopub.status.idle":"2022-07-14T03:38:42.830029Z","shell.execute_reply.started":"2022-07-14T03:38:42.815594Z","shell.execute_reply":"2022-07-14T03:38:42.829110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combine.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:42.831362Z","iopub.execute_input":"2022-07-14T03:38:42.831694Z","iopub.status.idle":"2022-07-14T03:38:42.852976Z","shell.execute_reply.started":"2022-07-14T03:38:42.831666Z","shell.execute_reply":"2022-07-14T03:38:42.851968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- 요일 전처리 (요일명을 숫자로)","metadata":{}},{"cell_type":"code","source":"combine.weekday = combine.weekday.astype('category')\n\nprint('실행 완료')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:42.854368Z","iopub.execute_input":"2022-07-14T03:38:42.854718Z","iopub.status.idle":"2022-07-14T03:38:42.866856Z","shell.execute_reply.started":"2022-07-14T03:38:42.854678Z","shell.execute_reply":"2022-07-14T03:38:42.865361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combine.weekday.cat.categories = ['5','1','6','0','4','2','3']\n\nprint('실행 완료')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:42.868312Z","iopub.execute_input":"2022-07-14T03:38:42.869040Z","iopub.status.idle":"2022-07-14T03:38:42.875405Z","shell.execute_reply.started":"2022-07-14T03:38:42.869005Z","shell.execute_reply":"2022-07-14T03:38:42.874486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- wind speed 데이터 결측치 채우기","metadata":{}},{"cell_type":"code","source":"dataWind0 = combine[combine['windspeed']==0]\ndataWindNot0 = combine[combine['windspeed']!=0]\n\nprint('실행 완료')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:42.883374Z","iopub.execute_input":"2022-07-14T03:38:42.883998Z","iopub.status.idle":"2022-07-14T03:38:42.897839Z","shell.execute_reply.started":"2022-07-14T03:38:42.883962Z","shell.execute_reply":"2022-07-14T03:38:42.896573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataWind0.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:42.899143Z","iopub.execute_input":"2022-07-14T03:38:42.900139Z","iopub.status.idle":"2022-07-14T03:38:42.908203Z","shell.execute_reply.started":"2022-07-14T03:38:42.900084Z","shell.execute_reply":"2022-07-14T03:38:42.907025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataWind0_df = dataWind0.drop(['windspeed','casual','registered','count','datetime','tempDate'],axis=1)\n\ndataWindNot0_df = dataWindNot0.drop(['windspeed','casual','registered','count','datetime','tempDate'],axis=1)\ndataWindNot0_series = dataWindNot0['windspeed']\n\nprint('실행 완료')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:42.909539Z","iopub.execute_input":"2022-07-14T03:38:42.910464Z","iopub.status.idle":"2022-07-14T03:38:42.922733Z","shell.execute_reply.started":"2022-07-14T03:38:42.910424Z","shell.execute_reply":"2022-07-14T03:38:42.921526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataWindNot0_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:42.924254Z","iopub.execute_input":"2022-07-14T03:38:42.925174Z","iopub.status.idle":"2022-07-14T03:38:42.945711Z","shell.execute_reply.started":"2022-07-14T03:38:42.925137Z","shell.execute_reply":"2022-07-14T03:38:42.944832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataWind0_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:42.947125Z","iopub.execute_input":"2022-07-14T03:38:42.947734Z","iopub.status.idle":"2022-07-14T03:38:42.968671Z","shell.execute_reply.started":"2022-07-14T03:38:42.947698Z","shell.execute_reply":"2022-07-14T03:38:42.967500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- wind speed가 0이 아닌 데이터를 활용하여 모델링을 통해 wind speed가 0인 결측치들을 채운다","metadata":{}},{"cell_type":"code","source":"rf2 = RandomForestRegressor()\nrf2.fit(dataWindNot0_df,dataWindNot0_series)\npredicted = rf2.predict(dataWind0_df)\nprint(predicted)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:42.970161Z","iopub.execute_input":"2022-07-14T03:38:42.970830Z","iopub.status.idle":"2022-07-14T03:38:48.548622Z","shell.execute_reply.started":"2022-07-14T03:38:42.970785Z","shell.execute_reply":"2022-07-14T03:38:48.547349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataWind0['windspeed'] = predicted","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:48.550171Z","iopub.execute_input":"2022-07-14T03:38:48.550527Z","iopub.status.idle":"2022-07-14T03:38:48.556445Z","shell.execute_reply.started":"2022-07-14T03:38:48.550498Z","shell.execute_reply":"2022-07-14T03:38:48.555024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combine = pd.concat([dataWind0,dataWindNot0],axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:48.557992Z","iopub.execute_input":"2022-07-14T03:38:48.558472Z","iopub.status.idle":"2022-07-14T03:38:48.576975Z","shell.execute_reply.started":"2022-07-14T03:38:48.558427Z","shell.execute_reply":"2022-07-14T03:38:48.576018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#우리가 가진 column들 중 값들이 일정하고 정해져있다면 category로 변경해주고\n#필요하지 않은 column들은 이제 버린다.\ncategorizational_columns = ['holiday','humidity','season','weather','workingday','year','month','day','hour']\ndrop_columns = ['datetime','casual','registered','count','tempDate']","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:48.578342Z","iopub.execute_input":"2022-07-14T03:38:48.579394Z","iopub.status.idle":"2022-07-14T03:38:48.584983Z","shell.execute_reply.started":"2022-07-14T03:38:48.579355Z","shell.execute_reply":"2022-07-14T03:38:48.583818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#categorical하게 변환\nfor col in categorizational_columns:\n    combine[col] = combine[col].astype('category')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:48.586337Z","iopub.execute_input":"2022-07-14T03:38:48.586649Z","iopub.status.idle":"2022-07-14T03:38:48.605872Z","shell.execute_reply.started":"2022-07-14T03:38:48.586622Z","shell.execute_reply":"2022-07-14T03:38:48.604440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#합쳐진 combine데이터 셋에서 count의 유무로 훈련과 테스트셋을 분리하고 각각을 datetime으로 정렬\ntrain = combine[pd.notnull(combine['count'])].sort_values(by='datetime')\ntest = combine[~pd.notnull(combine['count'])].sort_values(by='datetime')\n\n#데이터 훈련시 집어 넣게 될 각각의 결과 값들\ndatetimecol = test['datetime']\nyLabels = train['count'] #count\nyLabelsRegistered = train['registered'] # 회원\nyLabelsCasual = train['casual'] # 비회원","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:48.607279Z","iopub.execute_input":"2022-07-14T03:38:48.607632Z","iopub.status.idle":"2022-07-14T03:38:48.636331Z","shell.execute_reply.started":"2022-07-14T03:38:48.607600Z","shell.execute_reply":"2022-07-14T03:38:48.635071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#필요 없는 column들을 버린 후의 훈련과 테스트 셋\ntrain = train.drop(drop_columns,axis=1)\ntest = test.drop(drop_columns,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:48.637939Z","iopub.execute_input":"2022-07-14T03:38:48.638348Z","iopub.status.idle":"2022-07-14T03:38:48.646921Z","shell.execute_reply.started":"2022-07-14T03:38:48.638309Z","shell.execute_reply":"2022-07-14T03:38:48.645626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n해당 문제에서는 RMSLE방식을 이용하여 제대로 예측이 되었는지 평가하게 됨.\nRMSLE는 아래 링크를 참조하여 이용.\nhttps://programmers.co.kr/learn/courses/21/lessons/943#\n\nRMSLE\n과대평가 된 항목보다는 과소평가 된 항목에 페널티를 주는방식\n오차를 제곱하여 형균한 값의 제곱근으로 값이 작아질 수록 정밀도가 높음\n0에 가까운 값이 나올 수록 정밀도가 높다\n\"\"\"\n\n# y is predict value y_ is actual value\ndef rmsle(y,y_,convertExp=True):\n    if convertExp:\n        y = np.exp(y), \n        y_ = np.exp(y_)\n    log1 = np.nan_to_num(np.array([np.log(v + 1) for v in y]))\n    log2 = np.nan_to_num(np.array([np.log(v + 1) for v in y_]))\n    calc = (log1 - log2) ** 2\n    return np.sqrt(np.mean(calc))","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:48.648494Z","iopub.execute_input":"2022-07-14T03:38:48.648902Z","iopub.status.idle":"2022-07-14T03:38:48.658845Z","shell.execute_reply.started":"2022-07-14T03:38:48.648867Z","shell.execute_reply":"2022-07-14T03:38:48.657928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#선형 회귀 모델\n#선형 회귀모델은 건드릴 만한 내부 attr들이 없음\nfrom sklearn.linear_model import LinearRegression,Ridge,Lasso\n\n\nlr = LinearRegression()\n\n\"\"\"\n아래의 커널을 참조하여 yLabels를 로그화 하려는데 왜 np.log가 아닌 np.log1p를 활용하는가??\nnp.log1p는 np.log(1+x)와 동일. 이유는 만약 어떤 x값이 0인데 이를 log하게되면, (-)무한대로 수렴하기 때문에 np.log1p를 활용함. \n참조: https://ko.wikipedia.org/wiki/%EB%A1%9C%EA%B7%B8 \n\"\"\"\n\nyLabelslog = np.log1p(yLabels)\n#선형 모델에 우리의 데이터를 학습\nlr.fit(train,yLabelslog)\n#결과 값 도출\npreds = lr.predict(train)\n#rmsle함수의 element에 np.exp()지수 함수를 취하는 이유는 우리의 preds값에 얻어진 것은 한번 log를 한 값이기 때문에 원래 모델에는 log를 하지 않은 원래의 값을 넣기 위함임.\nprint('RMSLE Value For Linear Regression: {}'.format(rmsle(np.exp(yLabelslog),np.exp(preds),False)))","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:48.660279Z","iopub.execute_input":"2022-07-14T03:38:48.660680Z","iopub.status.idle":"2022-07-14T03:38:48.834307Z","shell.execute_reply.started":"2022-07-14T03:38:48.660644Z","shell.execute_reply":"2022-07-14T03:38:48.833413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n데이터 훈련시 Log값을 취하는 이유??\n우리가 결과 값으로 투입하는 Count값이 최저 값과 최고 값의 낙폭이 너무 커서\n만약 log를 취하지 않고 해보면 print하는 결과 값이 inf(infinity)로 뜨게 됨\n\"\"\"\n\n#count값의 분포\nsns.distplot(yLabels, bins=range(yLabels.min().astype('int'),yLabels.max().astype('int')))\n\n#기존 훈련 데이터셋의 count의 개수\nprint(yLabels.count()) #10886\n\n\"\"\" \n3 sigma를 활용한 이상치 확인\n참조 : https://ko.wikipedia.org/wiki/68-95-99.7_%EA%B7%9C%EC%B9%99\n\"\"\"\n#3시그마를 적용한 이상치를 배제한 훈련 데이터셋의 count의 개수\nyLabels[np.logical_and(yLabels.mean()-3*yLabels.std() <= yLabels,yLabels.mean()+3*yLabels.std() >= yLabels)].count() #10739\n#이상치들이 존재할 때는 log를 활용하여 값을 도출","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:48.836538Z","iopub.execute_input":"2022-07-14T03:38:48.836925Z","iopub.status.idle":"2022-07-14T03:38:50.556202Z","shell.execute_reply.started":"2022-07-14T03:38:48.836890Z","shell.execute_reply":"2022-07-14T03:38:50.554873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"GridSearchCV를 활용하면 우리가 이용하게 될 각각의 모델마다 변경해야 하는 파라미터 튜닝시 어떤 파라미터가 최적의 값을 내는지 등을 알 수 있음.\n\nGridSearchCV 참조:\nhttps://scikit-learn.org/stable/modules/generated/sklearn.model_selection.GridSearchCV.html\nhttps://datascienceschool.net/view-notebook/ff4b5d491cc34f94aea04baca86fbef8/","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV\nfrom sklearn import metrics\n\n#Ridge모델은 L2제약을 가지는 선형회귀모델에서 개선된 모델이며 해당 모델에서 유의 깊게 튜닝해야하는 파라미터는 alpha값이다.\nridge = Ridge()\n\n#우리가 튜닝하고자하는 Ridge의 파라미터 중 특정 파라미터에 배열 값으로 넘겨주게 되면 테스트 후 어떤 파라미터가 최적의 값인지 알려줌 \nridge_params = {'max_iter':[3000],'alpha':[0.001,0.01,0.1,1,10,100,1000]}\nrmsle_scorer = metrics.make_scorer(rmsle,greater_is_better=False)\ngrid_ridge = GridSearchCV(ridge,ridge_params,scoring=rmsle_scorer,cv=5)\n\ngrid_ridge.fit(train,yLabelslog)\npreds = grid_ridge.predict(train)\nprint(grid_ridge.best_params_)\nprint('RMSLE Value for Ridge Regression {}'.format(rmsle(np.exp(yLabelslog),np.exp(preds),False)))","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:50.558199Z","iopub.execute_input":"2022-07-14T03:38:50.558694Z","iopub.status.idle":"2022-07-14T03:38:52.444900Z","shell.execute_reply.started":"2022-07-14T03:38:50.558648Z","shell.execute_reply":"2022-07-14T03:38:52.443327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#결과에 대해 GridSearchCV의 변수인 grid_ridge변수에 cv_result_를 통해 alpha값의 변화에 따라 평균값의 변화를 파악 가능\ndf = pd.DataFrame(grid_ridge.cv_results_)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:52.446853Z","iopub.execute_input":"2022-07-14T03:38:52.447787Z","iopub.status.idle":"2022-07-14T03:38:52.463789Z","shell.execute_reply.started":"2022-07-14T03:38:52.447732Z","shell.execute_reply":"2022-07-14T03:38:52.461415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:52.466202Z","iopub.execute_input":"2022-07-14T03:38:52.467135Z","iopub.status.idle":"2022-07-14T03:38:52.508065Z","shell.execute_reply.started":"2022-07-14T03:38:52.467081Z","shell.execute_reply":"2022-07-14T03:38:52.506420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Ridge모델은 L1제약을 가지는 선형회귀모델에서 개선된 모델이며 해당 모델에서 유의 깊게 튜닝해야하는 파라미터는 alpha값이다.\nlasso = Lasso()\n\nlasso_params = {'max_iter':[3000],'alpha':[0.001,0.01,0.1,1,10,100,1000]}\ngrid_lasso = GridSearchCV(lasso,lasso_params,scoring=rmsle_scorer,cv=5)\ngrid_lasso.fit(train,yLabelslog)\npreds = grid_lasso.predict(train)\nprint('RMSLE Value for Lasso Regression {}'.format(rmsle(np.exp(yLabelslog),np.exp(preds),False)))","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:52.509964Z","iopub.execute_input":"2022-07-14T03:38:52.510993Z","iopub.status.idle":"2022-07-14T03:38:56.612676Z","shell.execute_reply.started":"2022-07-14T03:38:52.510952Z","shell.execute_reply":"2022-07-14T03:38:56.611374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf = RandomForestRegressor()\n\nrf_params = {'n_estimators':[1,10,100]}\ngrid_rf = GridSearchCV(rf,rf_params,scoring=rmsle_scorer,cv=5)\ngrid_rf.fit(train,yLabelslog)\npreds = grid_rf.predict(train)\nprint('RMSLE Value for RandomForest {}'.format(rmsle(np.exp(yLabelslog),np.exp(preds),False)))","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:38:56.614316Z","iopub.execute_input":"2022-07-14T03:38:56.614781Z","iopub.status.idle":"2022-07-14T03:39:20.624398Z","shell.execute_reply.started":"2022-07-14T03:38:56.614733Z","shell.execute_reply":"2022-07-14T03:39:20.623039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingRegressor\ngb= GradientBoostingRegressor()\ngb_params ={'max_depth':range(1,11,1), 'n_estimators': [1,10,100]}\ngrid_gb=GridSearchCV(gb,gb_params,scoring=rmsle_scorer,cv=5)\ngrid_gb.fit(train,yLabelslog)\npreds = grid_gb.predict(train)\nprint('RMSLE Value for GradientBoosting {}'.format(rmsle(np.exp(yLabelslog),np.exp(preds),False)))","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:39:20.626408Z","iopub.execute_input":"2022-07-14T03:39:20.626785Z","iopub.status.idle":"2022-07-14T03:40:52.656888Z","shell.execute_reply.started":"2022-07-14T03:39:20.626753Z","shell.execute_reply":"2022-07-14T03:40:52.655742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predsTest = grid_gb.predict(test)\nfig,(ax1,ax2)= plt.subplots(ncols=2)\nfig.set_size_inches(12,5)\nsns.distplot(yLabels,ax=ax1,bins=50)\nsns.distplot(np.exp(predsTest),ax=ax2,bins=50)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:40:52.658356Z","iopub.execute_input":"2022-07-14T03:40:52.659428Z","iopub.status.idle":"2022-07-14T03:40:53.184693Z","shell.execute_reply.started":"2022-07-14T03:40:52.659380Z","shell.execute_reply":"2022-07-14T03:40:53.183432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submisson = pd.DataFrame({\n    'datetime' : datetimecol,\n    'count' : [max(0,x) for x in np.exp(predsTest)]\n})\nsubmisson.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:40:53.186193Z","iopub.execute_input":"2022-07-14T03:40:53.187288Z","iopub.status.idle":"2022-07-14T03:40:53.219226Z","shell.execute_reply.started":"2022-07-14T03:40:53.187213Z","shell.execute_reply":"2022-07-14T03:40:53.218337Z"},"trusted":true},"execution_count":null,"outputs":[]}]}