{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## 프로젝트 개요 \n\n- 강의명 : 2022년 K-디지털 직업훈련(Training) 사업 - AI데이터플랫폼을 활용한 빅데이터 분석전문가 과정\n- 교과목명 : 빅데이터 분석 및 시각화, AI개발 기초, 인공지능 프로그래밍\n- 프로젝트 주제 : 캐글 대회 Bike Sharing Demand 데이터를 활용한 수요 예측 대회\n- 프로젝트 마감일 : 2022년 7월 19일 화요일\n- 수강생명 : 정상","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-13T03:11:11.382372Z","iopub.execute_input":"2022-07-13T03:11:11.383179Z","iopub.status.idle":"2022-07-13T03:11:11.395360Z","shell.execute_reply.started":"2022-07-13T03:11:11.383129Z","shell.execute_reply":"2022-07-13T03:11:11.393962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n\nimport numpy as np \nimport pandas as pd \nimport seaborn as sns \nimport matplotlib.pyplot as plt\nimport matplotlib \nimport calendar \nfrom scipy import stats\nimport missingno as msno\nfrom datetime import datetime\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.linear_model import LinearRegression,Ridge,Lasso\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn import metrics\nfrom sklearn.ensemble import GradientBoostingRegressor\nimport xgboost as xgb\nfrom xgboost import XGBClassifier\nfrom xgboost import XGBRegressor\nfrom sklearn.datasets import load_breast_cancer\nfrom sklearn.model_selection import train_test_split\n\n\nimport os","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:11.503571Z","iopub.execute_input":"2022-07-13T03:11:11.504477Z","iopub.status.idle":"2022-07-13T03:11:11.513480Z","shell.execute_reply.started":"2022-07-13T03:11:11.504426Z","shell.execute_reply":"2022-07-13T03:11:11.512607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 데이터 불러오기 ","metadata":{}},{"cell_type":"code","source":"DATA_PATH = '/kaggle/input/bike-sharing-demand/'\ntrain = pd.read_csv(DATA_PATH + 'train.csv')\ntest = pd.read_csv(DATA_PATH + 'test.csv')\nsubmission = pd.read_csv(DATA_PATH + 'sampleSubmission.csv')\ntrain.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:11.622162Z","iopub.execute_input":"2022-07-13T03:11:11.622880Z","iopub.status.idle":"2022-07-13T03:11:11.675665Z","shell.execute_reply.started":"2022-07-13T03:11:11.622845Z","shell.execute_reply":"2022-07-13T03:11:11.674756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = train.copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:11.755221Z","iopub.execute_input":"2022-07-13T03:11:11.756178Z","iopub.status.idle":"2022-07-13T03:11:11.762519Z","shell.execute_reply.started":"2022-07-13T03:11:11.756128Z","shell.execute_reply":"2022-07-13T03:11:11.761456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 결측치 유무 확인","metadata":{}},{"cell_type":"code","source":"msno.matrix(train, figsize=(12,5))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:11.870314Z","iopub.execute_input":"2022-07-13T03:11:11.871661Z","iopub.status.idle":"2022-07-13T03:11:12.188941Z","shell.execute_reply.started":"2022-07-13T03:11:11.871618Z","shell.execute_reply":"2022-07-13T03:11:12.187946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 이상치확인 ","metadata":{}},{"cell_type":"code","source":"# count자체와 season,holiday,workingday를 box플롯으로 시각화\nfig, ax=plt.subplots(nrows=2,ncols=2)\nfig.set_size_inches(20,15)\nsns.boxplot(y='count', data=train, orient='v', ax= ax[0,0])\nax[0,0].set_title('count')\nsns.boxplot(x='season',y='count', data=train, orient='v', ax= ax[0,1])\nax[0,1].set_title('count')\nsns.boxplot(x='holiday',y='count', data=train, orient='v', ax= ax[1,0])\nax[1,0].set_title('count')\nsns.boxplot(x='workingday',y='count', data=train, orient='v', ax= ax[1,1])\nax[1,1].set_title('count')","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:12.191114Z","iopub.execute_input":"2022-07-13T03:11:12.192303Z","iopub.status.idle":"2022-07-13T03:11:12.708251Z","shell.execute_reply.started":"2022-07-13T03:11:12.192253Z","shell.execute_reply":"2022-07-13T03:11:12.707338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 이상치 제거 ","metadata":{}},{"cell_type":"code","source":"# count 칼럼의 이상치 제거 \ntrainWithOutliers = train[np.abs(train[\"count\"]-train['count'].mean())<= (3*train['count'].std())]","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:12.709785Z","iopub.execute_input":"2022-07-13T03:11:12.710855Z","iopub.status.idle":"2022-07-13T03:11:12.721123Z","shell.execute_reply.started":"2022-07-13T03:11:12.710817Z","shell.execute_reply":"2022-07-13T03:11:12.719746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print (\"Shape Of The Before Ouliers: \",train.shape)\nprint (\"Shape Of The After Ouliers: \",trainWithOutliers.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:12.723500Z","iopub.execute_input":"2022-07-13T03:11:12.723921Z","iopub.status.idle":"2022-07-13T03:11:12.729934Z","shell.execute_reply.started":"2022-07-13T03:11:12.723869Z","shell.execute_reply":"2022-07-13T03:11:12.729023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 데이터 가공전 상관관계 분석 ","metadata":{}},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:12.731568Z","iopub.execute_input":"2022-07-13T03:11:12.732009Z","iopub.status.idle":"2022-07-13T03:11:12.753930Z","shell.execute_reply.started":"2022-07-13T03:11:12.731977Z","shell.execute_reply":"2022-07-13T03:11:12.752707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 각각의 컬럼별 상관관계 시각화\nCorrMat = train[['temp','atemp','casual','holiday','registered','humidity', 'windspeed','count']].corr()\nmask=np.array(CorrMat)\nmask[np.tril_indices_from(mask)] = False\nfig = plt.figure(figsize=[20,10])\nsns.heatmap(CorrMat,mask=mask,vmax=1,vmin=-0.4,annot=True,square=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:12.755772Z","iopub.execute_input":"2022-07-13T03:11:12.756105Z","iopub.status.idle":"2022-07-13T03:11:13.172485Z","shell.execute_reply.started":"2022-07-13T03:11:12.756074Z","shell.execute_reply":"2022-07-13T03:11:13.171329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 시간데이터 처리","metadata":{}},{"cell_type":"code","source":"# 시간데이터를 년,달,일,시,평일로 분리\nimport time\nimport datetime\ntrain['date']= pd.to_datetime(train['datetime'])\ntrain['year']= train['date'].dt.year\ntrain['month']=train['date'].dt.month\ntrain['day']=train['date'].dt.day\ntrain['hour']=train['date'].dt.hour\ntrain['weekday']=train['date'].dt.day_name()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:13.174309Z","iopub.execute_input":"2022-07-13T03:11:13.175048Z","iopub.status.idle":"2022-07-13T03:11:13.203142Z","shell.execute_reply.started":"2022-07-13T03:11:13.174999Z","shell.execute_reply":"2022-07-13T03:11:13.201983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:13.205328Z","iopub.execute_input":"2022-07-13T03:11:13.205682Z","iopub.status.idle":"2022-07-13T03:11:13.225823Z","shell.execute_reply.started":"2022-07-13T03:11:13.205649Z","shell.execute_reply":"2022-07-13T03:11:13.224570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 필요가 없어진 date 삭제","metadata":{}},{"cell_type":"code","source":"# date컬럼이 이제 필요없으니 삭제\ntrain = train.drop('date', axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:13.227354Z","iopub.execute_input":"2022-07-13T03:11:13.228292Z","iopub.status.idle":"2022-07-13T03:11:13.237616Z","shell.execute_reply.started":"2022-07-13T03:11:13.228244Z","shell.execute_reply":"2022-07-13T03:11:13.236349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### count값과 뽑아낸 컬럼들간의 관계 파악","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows=2,ncols=2,)\nfig.set_size_inches(12,10)\n\n\nsns.barplot(x='year',y='count',data=train,ax=ax[0,0])\nax[0,0].set_title('year')\nsns.barplot(x='month',y='count',data=train,ax=ax[0,1])\nax[0,1].set_title('month')\nsns.barplot(x='day',y='count',data=train,ax=ax[1,0])\nax[1,0].set_title('day')\nsns.barplot(x='hour',y='count',data=train,ax=ax[1,1])\nax[1,1].set_title('hour')","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:13.239380Z","iopub.execute_input":"2022-07-13T03:11:13.240023Z","iopub.status.idle":"2022-07-13T03:11:16.352078Z","shell.execute_reply.started":"2022-07-13T03:11:13.239984Z","shell.execute_reply":"2022-07-13T03:11:16.351142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 날씨, 계절, 작업일, 휴일과 count관계 파악","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows=2,ncols=2)\nfig.set_size_inches(12,10)\n\nsns.barplot(x='weather',y='count',data=train,ax=ax[0,0])\nax[0,0].set_title('weather')\nsns.barplot(x='season',y='count',data=train,ax=ax[0,1])\nax[0,1].set_title('season')\nsns.barplot(x='workingday',y='count',data=train,ax=ax[1,0])\nax[1,0].set_title('workingday')\nsns.barplot(x='holiday',y='count',data=train,ax=ax[1,1])\nax[1,1].set_title('holiday')","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:16.353332Z","iopub.execute_input":"2022-07-13T03:11:16.353875Z","iopub.status.idle":"2022-07-13T03:11:17.674312Z","shell.execute_reply.started":"2022-07-13T03:11:16.353842Z","shell.execute_reply":"2022-07-13T03:11:17.673049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def badToRight(month):\n    if month in [12,1,2]:\n        return 4\n    elif month in [3,4,5]:\n        return 1\n    elif month in [6,7,8]:\n        return 2\n    elif month in [9,10,11]:\n        return 3\n\ntrain['season'] = train.month.apply(badToRight)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:17.678394Z","iopub.execute_input":"2022-07-13T03:11:17.678799Z","iopub.status.idle":"2022-07-13T03:11:17.696139Z","shell.execute_reply.started":"2022-07-13T03:11:17.678766Z","shell.execute_reply":"2022-07-13T03:11:17.694488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows=2,ncols=2)\nfig.set_size_inches(12,10)\n\nsns.barplot(x='weather',y='count',data=train,ax=ax[0,0])\nax[0,0].set_title('weather')\nsns.barplot(x='season',y='count',data=train,ax=ax[0,1])\nax[0,1].set_title('season')\nsns.barplot(x='workingday',y='count',data=train,ax=ax[1,0])\nax[1,0].set_title('workingday')\nsns.barplot(x='holiday',y='count',data=train,ax=ax[1,1])\nax[1,1].set_title('holiday')","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:17.698022Z","iopub.execute_input":"2022-07-13T03:11:17.698424Z","iopub.status.idle":"2022-07-13T03:11:20.138702Z","shell.execute_reply.started":"2022-07-13T03:11:17.698386Z","shell.execute_reply":"2022-07-13T03:11:20.137534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 남은칼럼과 count와 비교","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows=2,ncols=2)\n\nfig.tight_layout()\nfig.set_size_inches(12,10)\n\nsns.regplot(x='temp', y='count', data = train, scatter_kws = {'alpha':0.2}, line_kws={'color':'blue'}, ax=ax[0,0])\nsns.regplot(x='atemp', y='count', data = train, scatter_kws = {'alpha':0.2}, line_kws={'color':'blue'}, ax=ax[0,1])\nsns.regplot(x='humidity', y='count', data = train, scatter_kws = {'alpha':0.2}, line_kws={'color':'blue'}, ax=ax[1,0])\nsns.regplot(x='windspeed', y='count', data = train, scatter_kws = {'alpha':0.2}, line_kws={'color':'blue'}, ax=ax[1,1])","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:20.140468Z","iopub.execute_input":"2022-07-13T03:11:20.140949Z","iopub.status.idle":"2022-07-13T03:11:23.862252Z","shell.execute_reply.started":"2022-07-13T03:11:20.140903Z","shell.execute_reply":"2022-07-13T03:11:23.861127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 상관계수 시각화","metadata":{}},{"cell_type":"code","source":"CorrMat = train.corr()\nmask = np.array(CorrMat)\nmask[np.tril_indices_from(mask)] = False\nfig = plt.figure(figsize=[20,10])\nsns.heatmap(CorrMat,mask=mask,vmin=-0.4,annot=True,square=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:23.863940Z","iopub.execute_input":"2022-07-13T03:11:23.864295Z","iopub.status.idle":"2022-07-13T03:11:24.717867Z","shell.execute_reply.started":"2022-07-13T03:11:23.864262Z","shell.execute_reply":"2022-07-13T03:11:24.716662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## count의 표준화","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows=2,ncols=2)\nfig.set_size_inches(10,6)\nsns.distplot(train['count'], ax=ax[0,0])\nstats.probplot(train['count'],dist='norm',fit=True,plot=ax[0,1])\nsns.distplot(np.log(trainWithOutliers['count']),ax=ax[1,0])\nstats.probplot(np.log(trainWithOutliers['count']),dist='norm',fit=True,plot=ax[1,1])","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:24.719216Z","iopub.execute_input":"2022-07-13T03:11:24.719621Z","iopub.status.idle":"2022-07-13T03:11:25.586228Z","shell.execute_reply.started":"2022-07-13T03:11:24.719587Z","shell.execute_reply":"2022-07-13T03:11:25.585077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 히트맵을 보고 두개의 컬럼 시각화 ","metadata":{}},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:25.587983Z","iopub.execute_input":"2022-07-13T03:11:25.588346Z","iopub.status.idle":"2022-07-13T03:11:25.604808Z","shell.execute_reply.started":"2022-07-13T03:11:25.588313Z","shell.execute_reply":"2022-07-13T03:11:25.603978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows = 5)\n\n## 전체 그래프 기본설정\n#  전체 그래프 사이즈 관리\nfig.set_size_inches(12,18)\n\n# 시간과 카운트에 대해 시각화\nsns.pointplot(x = 'hour', y ='count', hue = 'workingday', data = train, ax=ax[0])\nsns.pointplot(x = 'hour', y ='count', hue = 'holiday', data = train, ax=ax[1])\nsns.pointplot(x = 'hour', y ='count', hue = 'weekday', data = train, ax=ax[2])\nsns.pointplot(x = 'hour', y ='count', hue = 'season', data = train, ax=ax[3])\nsns.pointplot(x = 'hour', y ='count', hue = 'weather', data = train, ax=ax[4])","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:25.606051Z","iopub.execute_input":"2022-07-13T03:11:25.606737Z","iopub.status.idle":"2022-07-13T03:11:41.943677Z","shell.execute_reply.started":"2022-07-13T03:11:25.606680Z","shell.execute_reply":"2022-07-13T03:11:41.942503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- train의 weather값이 이상함을 발견","metadata":{}},{"cell_type":"code","source":"# 전체데이터중 weather=4인값 확\ntrain[train.weather == 4]","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:41.945250Z","iopub.execute_input":"2022-07-13T03:11:41.945856Z","iopub.status.idle":"2022-07-13T03:11:41.963867Z","shell.execute_reply.started":"2022-07-13T03:11:41.945822Z","shell.execute_reply":"2022-07-13T03:11:41.962790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows = 2)\n\n## 전체 그래프 기본설정\n#  전체 그래프 사이즈 관리\nfig.set_size_inches(12,18)\n\nsns.pointplot(x = 'month', y ='count', hue = 'weather', data = train, ax=ax[0])\nsns.barplot(x = 'month', y ='count', data = train, ax=ax[1])","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:41.965448Z","iopub.execute_input":"2022-07-13T03:11:41.965815Z","iopub.status.idle":"2022-07-13T03:11:44.272922Z","shell.execute_reply.started":"2022-07-13T03:11:41.965782Z","shell.execute_reply":"2022-07-13T03:11:44.271513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- windspeed는 0인값이 많은데 이는 0이었는지 측정하지못해서 0인지 두개의 경우가있다, 우리는 후자의 경우를 사용한다.","metadata":{}},{"cell_type":"code","source":"# 0 : Sunday -> 6: saturday\n# 머린러닝을 위해서 숫자로 변환해준다\ntrain['weekday']= train.weekday.astype('category')\ntrain.weekday.cat.categories =['5','1','6','0','4','2','3']","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:44.279240Z","iopub.execute_input":"2022-07-13T03:11:44.279653Z","iopub.status.idle":"2022-07-13T03:11:44.288642Z","shell.execute_reply.started":"2022-07-13T03:11:44.279619Z","shell.execute_reply":"2022-07-13T03:11:44.287690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### RandomForest로 Windspeed값 부여\n- 데이터를 windspeed == 0, windspeed != 0로 분리\n- 학습시킬 데이터 중 0이아닌 데이터프레임에서는 Windspeed만 담긴 Series와 학습시킬 column들의 데이터프레임으로 분리\n- 학습 후 Windspeed가 0인 데이터프레임에서 학습시킨 컬럼과 같게 추출해서 값을 부여받은 후, windspeed가 0인 데이터 프레임에 부여","metadata":{}},{"cell_type":"code","source":"# Windspeed = 0\nwindspeed_0 = train[train.windspeed == 0]\n# Windspeed != 0\nwindspeed_not0 = train[train.windspeed != 0]\n\n# windspeed=0인 프레임에 미포함\nwindspeed_0_df = windspeed_0.drop(['windspeed','casual','registered','count','datetime'], axis=1)\n\n# windspeed!=0인 데이터프렘인은 위와 동일하게 만들고, 학습시킬 Windspeed Series를 그대로 둠\nwindspeed_not0_df = windspeed_not0.drop(['windspeed','casual','registered','count','datetime'], axis=1)\nwindspeed_not0_series = windspeed_not0['windspeed']\n\n# 0이아닌 데이터프레임과 결과값인 시리즈 학습\nrf = RandomForestRegressor()\nrf.fit(windspeed_not0_df,windspeed_not0_series)\n\npredict_windspeed_0 = rf.predict(windspeed_0_df)\nwindspeed_0['windspeed'] = predict_windspeed_0","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:44.290217Z","iopub.execute_input":"2022-07-13T03:11:44.290652Z","iopub.status.idle":"2022-07-13T03:11:47.694382Z","shell.execute_reply.started":"2022-07-13T03:11:44.290533Z","shell.execute_reply":"2022-07-13T03:11:47.693199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# windspeed_0와 windspeed_not0를 다시 합치기\ntrain=pd.concat([windspeed_0,windspeed_not0],axis =0)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:47.695810Z","iopub.execute_input":"2022-07-13T03:11:47.696160Z","iopub.status.idle":"2022-07-13T03:11:47.706299Z","shell.execute_reply.started":"2022-07-13T03:11:47.696128Z","shell.execute_reply":"2022-07-13T03:11:47.705036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 합쳐진 데이터 datetime순 정렬\ntrain = train.sort_values(by=['datetime'])","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:47.709887Z","iopub.execute_input":"2022-07-13T03:11:47.711219Z","iopub.status.idle":"2022-07-13T03:11:47.731537Z","shell.execute_reply.started":"2022-07-13T03:11:47.711172Z","shell.execute_reply":"2022-07-13T03:11:47.730197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 바뀐 windspeed와 상관계수 분석 \nfig = plt.figure(figsize=[20,20])\nax = sns.heatmap(train.corr(),mask=mask,annot=True,square=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:47.733324Z","iopub.execute_input":"2022-07-13T03:11:47.734355Z","iopub.status.idle":"2022-07-13T03:11:48.603242Z","shell.execute_reply.started":"2022-07-13T03:11:47.734318Z","shell.execute_reply":"2022-07-13T03:11:48.602143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=[5,5])\nsns.distplot(train['windspeed'],bins=np.linspace(train['windspeed'].min(),train['windspeed'].max(),10))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:48.605203Z","iopub.execute_input":"2022-07-13T03:11:48.605612Z","iopub.status.idle":"2022-07-13T03:11:48.910740Z","shell.execute_reply.started":"2022-07-13T03:11:48.605569Z","shell.execute_reply":"2022-07-13T03:11:48.909933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 전처리 진행","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(DATA_PATH + 'train.csv')\ntest = pd.read_csv(DATA_PATH + 'test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:48.911979Z","iopub.execute_input":"2022-07-13T03:11:48.912952Z","iopub.status.idle":"2022-07-13T03:11:48.948871Z","shell.execute_reply.started":"2022-07-13T03:11:48.912915Z","shell.execute_reply":"2022-07-13T03:11:48.947602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 데이터를 합친상태에서 한번에 진행","metadata":{}},{"cell_type":"code","source":"combine = train.append(test)\ncombine.reset_index(inplace=True)\ncombine.drop('index',inplace=True,axis=1)\ncombine.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:48.950596Z","iopub.execute_input":"2022-07-13T03:11:48.950979Z","iopub.status.idle":"2022-07-13T03:11:48.978275Z","shell.execute_reply.started":"2022-07-13T03:11:48.950943Z","shell.execute_reply":"2022-07-13T03:11:48.977417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combine['date']= pd.to_datetime(combine['datetime'])\ncombine['year']= combine['date'].dt.year\ncombine['month']=combine['date'].dt.month\ncombine['day']=combine['date'].dt.day\ncombine['hour']=combine['date'].dt.hour\ncombine['weekday']=combine['date'].dt.day_name()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:48.979487Z","iopub.execute_input":"2022-07-13T03:11:48.980476Z","iopub.status.idle":"2022-07-13T03:11:49.012152Z","shell.execute_reply.started":"2022-07-13T03:11:48.980441Z","shell.execute_reply":"2022-07-13T03:11:49.011007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combine.weekday = combine.weekday.astype('category')","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:49.013375Z","iopub.execute_input":"2022-07-13T03:11:49.013687Z","iopub.status.idle":"2022-07-13T03:11:49.023581Z","shell.execute_reply.started":"2022-07-13T03:11:49.013658Z","shell.execute_reply":"2022-07-13T03:11:49.022524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combine.weekday.cat.categories = ['5','1','6','0','4','2','3']","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:49.025396Z","iopub.execute_input":"2022-07-13T03:11:49.026676Z","iopub.status.idle":"2022-07-13T03:11:49.033904Z","shell.execute_reply.started":"2022-07-13T03:11:49.026625Z","shell.execute_reply":"2022-07-13T03:11:49.032782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataWind0 = combine[combine[\"windspeed\"]==0]\ndataWindNot0 = combine[combine[\"windspeed\"]!=0]\nrfModel_wind = RandomForestRegressor()\nwindColumns = [\"season\",\"weather\",\"humidity\",\"month\",\"temp\",\"year\",\"atemp\"]\nrfModel_wind.fit(dataWindNot0[windColumns], dataWindNot0[\"windspeed\"])\n\nwind0Values = rfModel_wind.predict(X= dataWind0[windColumns])\ndataWind0[\"windspeed\"] = wind0Values\ncombine = dataWindNot0.append(dataWind0)\ncombine.reset_index(inplace=True)\ncombine.drop('index',inplace=True,axis=1)\ncombine.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:49.035681Z","iopub.execute_input":"2022-07-13T03:11:49.037014Z","iopub.status.idle":"2022-07-13T03:11:52.309034Z","shell.execute_reply.started":"2022-07-13T03:11:49.036888Z","shell.execute_reply":"2022-07-13T03:11:52.307744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combine = pd.concat([dataWind0,dataWindNot0], axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:52.310389Z","iopub.execute_input":"2022-07-13T03:11:52.310739Z","iopub.status.idle":"2022-07-13T03:11:52.320689Z","shell.execute_reply.started":"2022-07-13T03:11:52.310696Z","shell.execute_reply":"2022-07-13T03:11:52.319838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 인덱싱","metadata":{}},{"cell_type":"code","source":"# 우리가 가진 column중에 값이 일정하고 정해져 있으면 category로 변경\n# 필요하지 않은 column들은 버리기\ncategoricalFeatureNames = [\"season\",\"holiday\",\"workingday\",\"weather\",\"weekday\",\"month\",\"year\",\"hour\"]\nnumericalFeatureNames = [\"temp\",\"humidity\",\"windspeed\",\"atemp\"]\ndropFeatures = ['casual',\"count\",\"datetime\",\"date\",\"registered\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:52.322219Z","iopub.execute_input":"2022-07-13T03:11:52.322599Z","iopub.status.idle":"2022-07-13T03:11:52.330961Z","shell.execute_reply.started":"2022-07-13T03:11:52.322564Z","shell.execute_reply":"2022-07-13T03:11:52.329808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# categorical 하게 변환\nfor col in categoricalFeatureNames:\n    combine[col] = combine[col].astype('category')\n#for col in numericalFeatureNames:\n   # combine[col] = combine[col].astype('int64')","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:52.332346Z","iopub.execute_input":"2022-07-13T03:11:52.332943Z","iopub.status.idle":"2022-07-13T03:11:52.350900Z","shell.execute_reply.started":"2022-07-13T03:11:52.332906Z","shell.execute_reply":"2022-07-13T03:11:52.349662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 데이터 분리","metadata":{}},{"cell_type":"code","source":"train = combine[pd.notnull(combine['count'])].sort_values(by='datetime')\ntest = combine[~pd.notnull(combine['count'])].sort_values(by='datetime')\n\n#데이터 훈련시 집어 넣게 될 각각의 결과 값들\ndatetimecol = test['datetime']\nyLabels = train['count'] #count\nyLabelsRegistered = train['registered'] #등록된 사용자\nyLabelsCasual = train['casual'] #임시 사용자\ntrain.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:52.352528Z","iopub.execute_input":"2022-07-13T03:11:52.352906Z","iopub.status.idle":"2022-07-13T03:11:52.399702Z","shell.execute_reply.started":"2022-07-13T03:11:52.352869Z","shell.execute_reply":"2022-07-13T03:11:52.398782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 필요없는 컬럼 버리기\ntrain = train.drop(dropFeatures,axis=1)\ntest = test.drop(dropFeatures,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:52.401214Z","iopub.execute_input":"2022-07-13T03:11:52.402143Z","iopub.status.idle":"2022-07-13T03:11:52.409813Z","shell.execute_reply.started":"2022-07-13T03:11:52.402105Z","shell.execute_reply":"2022-07-13T03:11:52.408596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['weather'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:52.411149Z","iopub.execute_input":"2022-07-13T03:11:52.411495Z","iopub.status.idle":"2022-07-13T03:11:52.429171Z","shell.execute_reply.started":"2022-07-13T03:11:52.411464Z","shell.execute_reply":"2022-07-13T03:11:52.427711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## RMSLE 구하기\n- 오차를 제곱하여 평균한 값의 제곱근\n- 0에 가까울수록 정밀도가 높다","metadata":{}},{"cell_type":"code","source":"def rmsle(y, y_,convertExp=True):\n    if convertExp:\n        y = np.exp(y), \n        y_ = np.exp(y_)\n    log1 = np.nan_to_num(np.array([np.log(v + 1) for v in y]))\n    log2 = np.nan_to_num(np.array([np.log(v + 1) for v in y_]))\n    calc = (log1 - log2) ** 2\n    return np.sqrt(np.mean(calc))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:52.431151Z","iopub.execute_input":"2022-07-13T03:11:52.432461Z","iopub.status.idle":"2022-07-13T03:11:52.441434Z","shell.execute_reply.started":"2022-07-13T03:11:52.432406Z","shell.execute_reply":"2022-07-13T03:11:52.440238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"아래의 커널을 참조하여 yLabels를 로그화 하려는데 왜 np.log가 아닌 np.log1p를 활용하는가??\nnp.log1p는 np.log(1+x)와 동일. 이유는 만약 어떤 x값이 0인데 이를 log하게되면, (-)무한대로 수렴하기 때문에 np.log1p를 활용함. \n","metadata":{}},{"cell_type":"code","source":"# 선형회귀모델 사용\nlr = LinearRegression()\n\nyLabelslog = np.log1p(yLabels)\n#선형 모델에 우리의 데이터를 학습\nlr.fit(train,yLabelslog)\n#결과 값 도출\npreds = lr.predict(train)\n#rmsle함수의 element에 np.exp()지수 함수를 취하는 이유는 우리의 preds값에 얻어진 것은 한번 log를 한 값이기 때문에 원래 모델에는 log를 하지 않은 원래의 값을 넣기 위함임.\nprint('RMSLE Value For Linear Regression: {}'.format(rmsle(np.exp(yLabelslog),np.exp(preds),False)))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:52.443144Z","iopub.execute_input":"2022-07-13T03:11:52.443852Z","iopub.status.idle":"2022-07-13T03:11:52.635822Z","shell.execute_reply.started":"2022-07-13T03:11:52.443817Z","shell.execute_reply":"2022-07-13T03:11:52.634628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- 데이터 훈련시 Log값을 취하는 이유??\n- 우리가 결과 값으로 투입하는 Count값이 최저 값과 최고 값의 낙폭이 너무 커서\n- 만약 log를 취하지 않고 해보면 print하는 결과 값이 inf(infinity)로 뜨게 됨","metadata":{}},{"cell_type":"code","source":"# count값의 분포\nsns.distplot(yLabels,bins=range(yLabels.min().astype('int'),yLabels.max().astype('int')))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:52.637329Z","iopub.execute_input":"2022-07-13T03:11:52.638247Z","iopub.status.idle":"2022-07-13T03:11:54.547229Z","shell.execute_reply.started":"2022-07-13T03:11:52.638197Z","shell.execute_reply":"2022-07-13T03:11:54.545948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 기존 훈련 데이터 셋의 count 갯수 \nprint(yLabels.count())","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:54.549771Z","iopub.execute_input":"2022-07-13T03:11:54.550548Z","iopub.status.idle":"2022-07-13T03:11:54.556349Z","shell.execute_reply.started":"2022-07-13T03:11:54.550498Z","shell.execute_reply":"2022-07-13T03:11:54.555183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## GridSearchCV 사용","metadata":{}},{"cell_type":"code","source":"#Ridge모델은 L2제약을 가지는 선형회귀모델에서 개선된 모델이며 해당 모델에서 유의 깊게 튜닝해야하는 파라미터는 alpha값이다.\nridge = Ridge()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:54.557683Z","iopub.execute_input":"2022-07-13T03:11:54.558036Z","iopub.status.idle":"2022-07-13T03:11:54.568122Z","shell.execute_reply.started":"2022-07-13T03:11:54.558006Z","shell.execute_reply":"2022-07-13T03:11:54.566984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ridge_params = {'max_iter':[3000],'alpha':[0.1, 1, 2, 3, 4, 10, 30,100,200,300,400,800,900,1000]}\nrmsle_scorer = metrics.make_scorer(rmsle,greater_is_better=False)\ngrid_ridge = GridSearchCV(ridge,ridge_params,scoring=rmsle_scorer,cv=5)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:54.569607Z","iopub.execute_input":"2022-07-13T03:11:54.570507Z","iopub.status.idle":"2022-07-13T03:11:54.580929Z","shell.execute_reply.started":"2022-07-13T03:11:54.570454Z","shell.execute_reply":"2022-07-13T03:11:54.579731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid_ridge.fit(train,yLabelslog)\npreds = grid_ridge.predict(train)\nprint(grid_ridge.best_params_)\nprint('RMSLE Value for Ridge Regression {}'.format(rmsle(np.exp(yLabelslog),np.exp(preds),False)))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:54.582293Z","iopub.execute_input":"2022-07-13T03:11:54.582618Z","iopub.status.idle":"2022-07-13T03:11:59.294163Z","shell.execute_reply.started":"2022-07-13T03:11:54.582587Z","shell.execute_reply":"2022-07-13T03:11:59.293169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# GridSearchCV의 변수 grid_ridge에 cv_result_를 통해 alpha값의 변화에 따라 평균값의 변화를 파악\ndf = pd.DataFrame(grid_ridge.cv_results_)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:59.297791Z","iopub.execute_input":"2022-07-13T03:11:59.298260Z","iopub.status.idle":"2022-07-13T03:11:59.329122Z","shell.execute_reply.started":"2022-07-13T03:11:59.298216Z","shell.execute_reply":"2022-07-13T03:11:59.328125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Ridge모델은 선형회귀모델에서 개선된 모델이며 유의 깊게 튜닝해야하는 파라미터는 alpha값이다.\nlasso = Lasso()\n\nalpha  = 1/np.array([0.1, 1, 2, 3, 4, 10, 30,100,200,300,400,800,900,1000])\nlasso_params = {'max_iter':[3000],'alpha':alpha}\ngrid_lasso = GridSearchCV(lasso,lasso_params,scoring=rmsle_scorer,cv=5)\ngrid_lasso.fit(train,yLabelslog)\npreds = grid_lasso.predict(train)\nprint (grid_lasso.best_params_)\nprint('RMSLE Value for Lasso Regression {}'.format(rmsle(np.exp(yLabelslog),np.exp(preds),False)))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:11:59.330192Z","iopub.execute_input":"2022-07-13T03:11:59.330712Z","iopub.status.idle":"2022-07-13T03:12:08.309519Z","shell.execute_reply.started":"2022-07-13T03:11:59.330680Z","shell.execute_reply":"2022-07-13T03:12:08.308183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf = RandomForestRegressor()\n\nrf_params = {'n_estimators':[1,10,100]}\ngrid_rf = GridSearchCV(rf,rf_params,scoring=rmsle_scorer,cv=5)\ngrid_rf.fit(train,yLabelslog)\npreds = grid_rf.predict(train)\nprint('RMSLE Value for RandomForest {}'.format(rmsle(np.exp(yLabelslog),np.exp(preds),False)))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:12:08.317203Z","iopub.execute_input":"2022-07-13T03:12:08.317588Z","iopub.status.idle":"2022-07-13T03:12:33.300682Z","shell.execute_reply.started":"2022-07-13T03:12:08.317552Z","shell.execute_reply":"2022-07-13T03:12:33.299352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gb = GradientBoostingRegressor()\ngb_params={'max_depth':range(1,11,1),'n_estimators':[1,10,100]}\ngrid_gb=GridSearchCV(gb,gb_params,scoring=rmsle_scorer,cv=5)\ngrid_gb.fit(train,yLabelslog)\npreds = grid_gb.predict(train)\nprint('RMSLE Value for GradientBoosting {}'.format(rmsle(np.exp(yLabelslog),np.exp(preds),False)))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:12:33.302408Z","iopub.execute_input":"2022-07-13T03:12:33.303641Z","iopub.status.idle":"2022-07-13T03:14:09.127523Z","shell.execute_reply.started":"2022-07-13T03:12:33.303588Z","shell.execute_reply":"2022-07-13T03:14:09.126318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test.info()\npredsTest = grid_gb.predict(test)\nfig,(ax1,ax2)= plt.subplots(ncols=2)\nfig.set_size_inches(12,5)\nsns.distplot(yLabels,ax=ax1,bins=50)\nsns.distplot(np.exp(predsTest),ax=ax2,bins=50)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:14:09.129161Z","iopub.execute_input":"2022-07-13T03:14:09.129678Z","iopub.status.idle":"2022-07-13T03:14:09.802391Z","shell.execute_reply.started":"2022-07-13T03:14:09.129634Z","shell.execute_reply":"2022-07-13T03:14:09.800545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_preds = [max(0, x) for x in np.exp(predsTest)]\nsubmission['count'] = final_preds \nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:14:09.804471Z","iopub.execute_input":"2022-07-13T03:14:09.804855Z","iopub.status.idle":"2022-07-13T03:14:09.846296Z","shell.execute_reply.started":"2022-07-13T03:14:09.804820Z","shell.execute_reply":"2022-07-13T03:14:09.845251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:14:09.847651Z","iopub.execute_input":"2022-07-13T03:14:09.848192Z","iopub.status.idle":"2022-07-13T03:14:09.860278Z","shell.execute_reply.started":"2022-07-13T03:14:09.848160Z","shell.execute_reply":"2022-07-13T03:14:09.858822Z"},"trusted":true},"execution_count":null,"outputs":[]}]}