{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\n## 강의명 : 2022년 K-디지털 직업훈련(Training) \n- 사업 - AI데이터플랫폼을 활용한 빅데이터 분석전문가 과정\n- 교과목명 : 빅데이터 분석 및 시각화, AI개발 기초, 인공지능 프로그래밍\n- 프로젝트 주제 : 캐글 대회 Bike Sharing Demand 데이터를 활용한 수요 예측 대회\n- 프로젝트 마감일 : 2022년 7월 19일 화요일\n- 수강생명 : 김선형","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-12T01:31:50.393033Z","iopub.execute_input":"2022-07-12T01:31:50.393453Z","iopub.status.idle":"2022-07-12T01:31:50.400891Z","shell.execute_reply.started":"2022-07-12T01:31:50.393419Z","shell.execute_reply":"2022-07-12T01:31:50.400147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## STEP 01. 필요한 라이브러리를 호출한다","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport calendar\nfrom datetime import datetime\n\nimport os\nprint(os.listdir(\"../input\"))\nprint(\"pandas version:\",pd.__version__)\nprint(\"numpy version:\",np.__version__)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:50.487499Z","iopub.execute_input":"2022-07-12T01:31:50.488141Z","iopub.status.idle":"2022-07-12T01:31:50.494901Z","shell.execute_reply.started":"2022-07-12T01:31:50.488107Z","shell.execute_reply":"2022-07-12T01:31:50.493847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## STEP 02 데이터를 불러온다\n- 데이터를 불러온 후 shape, info()를 통해 데이터를 확인한다.","metadata":{}},{"cell_type":"code","source":"DATA_PATH = \"/kaggle/input/bike-sharing-demand/\"\ntrain = pd.read_csv(DATA_PATH + 'train.csv')\ntest = pd.read_csv(DATA_PATH + 'test.csv')\nsubmission = pd.read_csv(DATA_PATH +'sampleSubmission.csv')\n\ntrain.shape, test.shape, submission.shape\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:50.576857Z","iopub.execute_input":"2022-07-12T01:31:50.577247Z","iopub.status.idle":"2022-07-12T01:31:50.625829Z","shell.execute_reply.started":"2022-07-12T01:31:50.577215Z","shell.execute_reply":"2022-07-12T01:31:50.624949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:50.666446Z","iopub.execute_input":"2022-07-12T01:31:50.667240Z","iopub.status.idle":"2022-07-12T01:31:50.682266Z","shell.execute_reply.started":"2022-07-12T01:31:50.667189Z","shell.execute_reply":"2022-07-12T01:31:50.681414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:50.801836Z","iopub.execute_input":"2022-07-12T01:31:50.802844Z","iopub.status.idle":"2022-07-12T01:31:50.816409Z","shell.execute_reply.started":"2022-07-12T01:31:50.802794Z","shell.execute_reply":"2022-07-12T01:31:50.815667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- head()를 활용하여 train, test의 형태를 출력함.","metadata":{}},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:50.894828Z","iopub.execute_input":"2022-07-12T01:31:50.895418Z","iopub.status.idle":"2022-07-12T01:31:50.910157Z","shell.execute_reply.started":"2022-07-12T01:31:50.895386Z","shell.execute_reply":"2022-07-12T01:31:50.909300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:50.983667Z","iopub.execute_input":"2022-07-12T01:31:50.984255Z","iopub.status.idle":"2022-07-12T01:31:50.997764Z","shell.execute_reply.started":"2022-07-12T01:31:50.984225Z","shell.execute_reply":"2022-07-12T01:31:50.996998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- 탐색적 자료 분석을 위해 train 데이터를 복제함.","metadata":{}},{"cell_type":"code","source":"temp_df = train.copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:51.067791Z","iopub.execute_input":"2022-07-12T01:31:51.068196Z","iopub.status.idle":"2022-07-12T01:31:51.074207Z","shell.execute_reply.started":"2022-07-12T01:31:51.068164Z","shell.execute_reply":"2022-07-12T01:31:51.072792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## STEP 03. 데이터 전처리 및 시각화\n- split 함수를 활용하여 년, 월, 일, 시간을 분리","metadata":{}},{"cell_type":"code","source":"train['Date']= train.datetime.apply(lambda x:x.split())\ntrain['Date'] # 날짜와 시간을 선 분리","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:51.152569Z","iopub.execute_input":"2022-07-12T01:31:51.153354Z","iopub.status.idle":"2022-07-12T01:31:51.171044Z","shell.execute_reply.started":"2022-07-12T01:31:51.153318Z","shell.execute_reply":"2022-07-12T01:31:51.169947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- 날짜에서 년, 월, 일을 분리","metadata":{}},{"cell_type":"code","source":"train['year'] = train.Date.apply(lambda x:x[0].split('-')[0])\ntrain['month'] = train.Date.apply(lambda x:x[0].split('-')[1])\ntrain['day'] = train.Date.apply(lambda x:x[0].split('-')[2])\ntrain['year'][0]\ntrain['month'][100]\ntrain['day'][1000]","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:51.240990Z","iopub.execute_input":"2022-07-12T01:31:51.241426Z","iopub.status.idle":"2022-07-12T01:31:51.271699Z","shell.execute_reply.started":"2022-07-12T01:31:51.241392Z","shell.execute_reply":"2022-07-12T01:31:51.270615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- 날짜와 시간 다루기\n- DB를 만지다보면 날짜관련 데이터를 계산 혹은 수집해야할때가 있음.\n- 이때 사용하는 함수가 strptime함수사용법은 다음과같음.\n- 5번째 라인의 strptime함수 두번째 인자는 다음과 같은 형식으로 넣어야함.\n- 예를들어서 DB 안에 2018-08-01이라는 string date자료가 들어있을경우두번째 인자에는 '%Y-%m-%d' 로 해줘야함.\n## 출처: https://jaeyung1001.tistory.com/56 [공부방 & 일상:티스토리]","metadata":{}},{"cell_type":"code","source":"train['weekday'] = train.Date.apply(lambda x : calendar.day_name[datetime.strptime(x[0],\"%Y-%m-%d\").weekday()])\ntrain['weekday']","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:51.328022Z","iopub.execute_input":"2022-07-12T01:31:51.328591Z","iopub.status.idle":"2022-07-12T01:31:51.525995Z","shell.execute_reply.started":"2022-07-12T01:31:51.328560Z","shell.execute_reply":"2022-07-12T01:31:51.524860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- 시간을 분리함.","metadata":{}},{"cell_type":"code","source":"train['hour'] = train.Date.apply(lambda x:x[1].split(':')[0])\ntrain['hour'] # 데이터에 분과 초는 모두 00으로 표시되어 따로 분과 초 데이터를 생성하지 않음.","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:51.527676Z","iopub.execute_input":"2022-07-12T01:31:51.528012Z","iopub.status.idle":"2022-07-12T01:31:51.544025Z","shell.execute_reply.started":"2022-07-12T01:31:51.527982Z","shell.execute_reply":"2022-07-12T01:31:51.542898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Date 분리를 위해 사용된 컬럼을 삭제함.","metadata":{}},{"cell_type":"code","source":"train = train.drop('Date', axis =1)\ntrain.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:51.545434Z","iopub.execute_input":"2022-07-12T01:31:51.546498Z","iopub.status.idle":"2022-07-12T01:31:51.573374Z","shell.execute_reply.started":"2022-07-12T01:31:51.546443Z","shell.execute_reply":"2022-07-12T01:31:51.572315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:51.588190Z","iopub.execute_input":"2022-07-12T01:31:51.588525Z","iopub.status.idle":"2022-07-12T01:31:51.609118Z","shell.execute_reply.started":"2022-07-12T01:31:51.588498Z","shell.execute_reply":"2022-07-12T01:31:51.608125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"분리한 데이터 중, 문자열로 되어있는 데이터 들을 숫자형으로 만들 필요가 있음.","metadata":{}},{"cell_type":"code","source":"train['year']=pd.to_numeric(train.year,errors = 'coerce')\ntrain['month']=pd.to_numeric(train.month,errors = 'coerce')\ntrain['day']=pd.to_numeric(train.day,errors = 'coerce')\ntrain['hour']=pd.to_numeric(train.hour,errors = 'coerce')\ntrain.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:51.673875Z","iopub.execute_input":"2022-07-12T01:31:51.674271Z","iopub.status.idle":"2022-07-12T01:31:51.740794Z","shell.execute_reply.started":"2022-07-12T01:31:51.674240Z","shell.execute_reply":"2022-07-12T01:31:51.739617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows=1, ncols = 2)\n\nsns.histplot(train['count'], ax=ax[0])\nsns.histplot(np.log(train['count']),ax=ax[1])\n\nax[0].set_title('Normal Graph')\nax[1].set_title('Log Transformed Graph')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:51.802898Z","iopub.execute_input":"2022-07-12T01:31:51.803484Z","iopub.status.idle":"2022-07-12T01:31:52.298967Z","shell.execute_reply.started":"2022-07-12T01:31:51.803452Z","shell.execute_reply":"2022-07-12T01:31:52.297871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows = 2, ncols = 2)\n\n## 1단계 : 전체 그래프 기본 설정\n# 그래프 사이 간격\nfig.tight_layout()\n\n# 전체 그래프 사이즈 관리\nfig.set_size_inches(10, 9)\n\n## 2단계 :  각 개별 그래프 입력\nsns.barplot(x = 'year', y = 'count', data = train, ci = None, ax=ax[0,0])\nsns.barplot(x = 'month',y = 'count', data = train, ci = None, ax=ax[0,1])\nsns.barplot(x = 'day', y = 'count', data = train, ci = None, ax=ax[1,0])\nsns.barplot(x = 'hour', y = 'count', data = train, ci = None, ax=ax[1,1])\n\n## 3단계 : 디테일 옵션\nax[0, 0].set_title(\"Rental Amounts by Year\")\nax[0, 1].set_title(\"Rental Amounts by month\")\nax[1, 0].set_title(\"Rental Amounts by day\")\nax[1, 1].set_title(\"Rental Amounts by hour\")","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:52.301002Z","iopub.execute_input":"2022-07-12T01:31:52.301332Z","iopub.status.idle":"2022-07-12T01:31:53.182483Z","shell.execute_reply.started":"2022-07-12T01:31:52.301301Z","shell.execute_reply":"2022-07-12T01:31:53.181303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows = 2, ncols = 2)\n\nfig.set_size_inches(10,9)\n\nsns.barplot(x='season', y='count', data = train, ci = None , ax=ax[0,0])\nsns.barplot(x='weather', y='count', data = train, ci = None , ax=ax[0,1])\nsns.barplot(x='holiday', y='count', data = train, ci = None , ax=ax[1,0])\nsns.barplot(x='workingday', y='count', data = train, ci = None , ax=ax[1,1])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:53.183916Z","iopub.execute_input":"2022-07-12T01:31:53.184251Z","iopub.status.idle":"2022-07-12T01:31:53.630690Z","shell.execute_reply.started":"2022-07-12T01:31:53.184220Z","shell.execute_reply":"2022-07-12T01:31:53.629575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def badToRight(month):\n    if month in [12,1,2]:\n        return 4\n    elif month in [3,4,5]:\n        return 1\n    elif month in [6,7,8]:\n        return 2\n    elif month in [9,10,11]:\n        return 3\n    \ntrain['season'] = train.month.apply(badToRight)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:53.633611Z","iopub.execute_input":"2022-07-12T01:31:53.634733Z","iopub.status.idle":"2022-07-12T01:31:53.650959Z","shell.execute_reply.started":"2022-07-12T01:31:53.634667Z","shell.execute_reply":"2022-07-12T01:31:53.649792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows = 2, ncols = 2)\n\nfig.set_size_inches(10,9)\n\nsns.barplot(x='season', y='count', data = train, ci = None , ax=ax[0,0])\nsns.barplot(x='weather', y='count', data = train, ci = None , ax=ax[0,1])\nsns.barplot(x='holiday', y='count', data = train, ci = None , ax=ax[1,0])\nsns.barplot(x='workingday', y='count', data = train, ci = None , ax=ax[1,1])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:53.652246Z","iopub.execute_input":"2022-07-12T01:31:53.652569Z","iopub.status.idle":"2022-07-12T01:31:54.104940Z","shell.execute_reply.started":"2022-07-12T01:31:53.652539Z","shell.execute_reply":"2022-07-12T01:31:54.103737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- 수치가 다양한 컬럼들을 distplot을 활용하여 count와 비교","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows = 2, ncols = 2)\n\nfig.set_size_inches(10,9)\n\nsns.distplot(train.temp, kde = True, kde_kws = {\"color\":\"r\",\"alpha\":0.3,\"linewidth\":5, \"shade\":True},ax=ax[0,0])\nsns.distplot(train.atemp, kde = True, kde_kws = {\"color\":\"y\",\"alpha\":0.3,\"linewidth\":5, \"shade\":True},ax=ax[0,1])\nsns.distplot(train.humidity, kde = True, kde_kws = {\"color\":\"g\",\"alpha\":0.3,\"linewidth\":5, \"shade\":True},ax=ax[1,0])\nsns.distplot(train.windspeed, kde = True, kde_kws = {\"color\":\"b\",\"alpha\":0.3,\"linewidth\":5, \"shade\":True},ax=ax[1,1])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:54.106209Z","iopub.execute_input":"2022-07-12T01:31:54.106503Z","iopub.status.idle":"2022-07-12T01:31:55.253842Z","shell.execute_reply.started":"2022-07-12T01:31:54.106475Z","shell.execute_reply":"2022-07-12T01:31:55.252983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"각각의 컬럼들 간의 상관계수를 heatmap을 통해 시각화","metadata":{}},{"cell_type":"code","source":"fig = plt.figure(figsize=[20,20])\nax = sns.heatmap(train.corr(),annot=True,square=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:55.254886Z","iopub.execute_input":"2022-07-12T01:31:55.255788Z","iopub.status.idle":"2022-07-12T01:31:56.446782Z","shell.execute_reply.started":"2022-07-12T01:31:55.255751Z","shell.execute_reply":"2022-07-12T01:31:56.445748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- heatmap 상관관계를 참조 후 이전의 시각화와는 달리 두개의 서로 다른 컬럼이 적용된 count를 시각화","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows = 2, ncols = 2)\n\nfig.set_size_inches(10,9)\nsns.pointplot(x='hour', y='count', hue='season', data=train, ci = None, ax=ax[0,0])\nsns.pointplot(x='hour', y='count', hue='holiday', data=train, ci = None, ax=ax[0,1])\nsns.pointplot(x='hour', y='count', hue='weekday', data=train, ci = None, ax=ax[1,0])\nsns.pointplot(x='hour', y='count', hue='weather', data=train, ci = None, ax=ax[1,1])\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:56.448170Z","iopub.execute_input":"2022-07-12T01:31:56.448468Z","iopub.status.idle":"2022-07-12T01:31:57.944166Z","shell.execute_reply.started":"2022-07-12T01:31:56.448441Z","shell.execute_reply":"2022-07-12T01:31:57.943059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- 4 weather에서 이상치 확인","metadata":{}},{"cell_type":"code","source":"train[train.weather==4]","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:57.945794Z","iopub.execute_input":"2022-07-12T01:31:57.946108Z","iopub.status.idle":"2022-07-12T01:31:57.963319Z","shell.execute_reply.started":"2022-07-12T01:31:57.946079Z","shell.execute_reply":"2022-07-12T01:31:57.962160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- 달과 날씨에 따른 count","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows = 2)\nsns.pointplot(x='month', y= 'count', hue = 'weather', data = train, ci = None, ax = ax[0])\nsns.barplot(x = 'month', y= 'count', data = train, ci = None, ax = ax[1])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:57.967978Z","iopub.execute_input":"2022-07-12T01:31:57.968298Z","iopub.status.idle":"2022-07-12T01:31:58.415419Z","shell.execute_reply.started":"2022-07-12T01:31:57.968269Z","shell.execute_reply":"2022-07-12T01:31:58.414259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- windspeed 가 제대로 측정되지 못했다고 생각 한 후 데이터를 활용하여 값을 부여 해야함.\n\n- 머신러닝 모델에 훈련 시킬 때 문자열 값은 불가능하기 때문에 문자열을 카테고리화 하고 숫자로 변환할 필요가 있음","metadata":{}},{"cell_type":"code","source":"train['weekday'] = train.weekday.astype('category')\nprint(train['weekday'].cat.categories)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:58.416587Z","iopub.execute_input":"2022-07-12T01:31:58.417526Z","iopub.status.idle":"2022-07-12T01:31:58.425355Z","shell.execute_reply.started":"2022-07-12T01:31:58.417491Z","shell.execute_reply":"2022-07-12T01:31:58.424601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.weekday.cat.categories = ['5','1','6','0','4','2','3']\nprint(train['weekday'].cat.categories)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:58.426329Z","iopub.execute_input":"2022-07-12T01:31:58.427166Z","iopub.status.idle":"2022-07-12T01:31:58.437505Z","shell.execute_reply.started":"2022-07-12T01:31:58.427134Z","shell.execute_reply":"2022-07-12T01:31:58.436769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.datetime = pd.to_datetime(train.datetime, errors='coerce')\ntrain = train.sort_values(by=['datetime'])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:58.438427Z","iopub.execute_input":"2022-07-12T01:31:58.439150Z","iopub.status.idle":"2022-07-12T01:31:58.460015Z","shell.execute_reply.started":"2022-07-12T01:31:58.439109Z","shell.execute_reply":"2022-07-12T01:31:58.459052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- train, test set을 합쳐서 진행","metadata":{}},{"cell_type":"code","source":"DATA_PATH = \"/kaggle/input/bike-sharing-demand/\"\ntrain = pd.read_csv(DATA_PATH + 'train.csv')\ntest = pd.read_csv(DATA_PATH + 'test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:58.461814Z","iopub.execute_input":"2022-07-12T01:31:58.462776Z","iopub.status.idle":"2022-07-12T01:31:58.501459Z","shell.execute_reply.started":"2022-07-12T01:31:58.462703Z","shell.execute_reply":"2022-07-12T01:31:58.500129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_dt = pd.concat([train,test],axis=0)\nall_dt.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:58.503385Z","iopub.execute_input":"2022-07-12T01:31:58.503867Z","iopub.status.idle":"2022-07-12T01:31:58.526431Z","shell.execute_reply.started":"2022-07-12T01:31:58.503824Z","shell.execute_reply":"2022-07-12T01:31:58.525387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_dt['tempDate'] = all_dt.datetime.apply(lambda x:x.split())\nall_dt['weekday'] = all_dt.tempDate.apply(lambda x: calendar.day_name[datetime.strptime(x[0],\"%Y-%m-%d\").weekday()])\nall_dt['year'] = all_dt.tempDate.apply(lambda x: x[0].split('-')[0])\nall_dt['month'] = all_dt.tempDate.apply(lambda x: x[0].split('-')[1])\nall_dt['day'] = all_dt.tempDate.apply(lambda x: x[0].split('-')[2])\nall_dt['hour'] = all_dt.tempDate.apply(lambda x: x[1].split(':')[0])\nall_dt['year'] = pd.to_numeric(all_dt.year,errors='coerce')\nall_dt['month'] = pd.to_numeric(all_dt.month,errors='coerce')\nall_dt['day'] = pd.to_numeric(all_dt.day,errors='coerce')\nall_dt['hour'] = pd.to_numeric(all_dt.hour,errors='coerce')\nall_dt.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:58.527887Z","iopub.execute_input":"2022-07-12T01:31:58.528615Z","iopub.status.idle":"2022-07-12T01:31:59.054773Z","shell.execute_reply.started":"2022-07-12T01:31:58.528569Z","shell.execute_reply":"2022-07-12T01:31:59.053384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_dt.weekday = all_dt.weekday.astype('category')\nall_dt.weekday.cat.categories = ['5','1','6','0','4','2','3']","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:59.056438Z","iopub.execute_input":"2022-07-12T01:31:59.057275Z","iopub.status.idle":"2022-07-12T01:31:59.070923Z","shell.execute_reply.started":"2022-07-12T01:31:59.057230Z","shell.execute_reply":"2022-07-12T01:31:59.069891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_dt.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:59.074377Z","iopub.execute_input":"2022-07-12T01:31:59.078205Z","iopub.status.idle":"2022-07-12T01:31:59.101427Z","shell.execute_reply.started":"2022-07-12T01:31:59.078160Z","shell.execute_reply":"2022-07-12T01:31:59.099569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\ndataWind0 = all_dt[all_dt['windspeed']==0]\ndataWindNot0 = all_dt[all_dt['windspeed']!=0]\ndataWind0.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:59.103663Z","iopub.execute_input":"2022-07-12T01:31:59.107059Z","iopub.status.idle":"2022-07-12T01:31:59.123246Z","shell.execute_reply.started":"2022-07-12T01:31:59.107009Z","shell.execute_reply":"2022-07-12T01:31:59.122248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataWind0_df = dataWind0.drop(['windspeed','casual','registered','count','datetime','tempDate'],axis=1)\n\ndataWindNot0_df = dataWindNot0.drop(['windspeed','casual','registered','count','datetime','tempDate'],axis=1)\ndataWindNot0_series = dataWindNot0['windspeed']","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:59.124322Z","iopub.execute_input":"2022-07-12T01:31:59.125133Z","iopub.status.idle":"2022-07-12T01:31:59.134100Z","shell.execute_reply.started":"2022-07-12T01:31:59.125097Z","shell.execute_reply":"2022-07-12T01:31:59.132787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataWindNot0_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:59.135995Z","iopub.execute_input":"2022-07-12T01:31:59.136850Z","iopub.status.idle":"2022-07-12T01:31:59.158388Z","shell.execute_reply.started":"2022-07-12T01:31:59.136804Z","shell.execute_reply":"2022-07-12T01:31:59.157473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataWind0_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:59.159669Z","iopub.execute_input":"2022-07-12T01:31:59.160035Z","iopub.status.idle":"2022-07-12T01:31:59.177371Z","shell.execute_reply.started":"2022-07-12T01:31:59.160005Z","shell.execute_reply":"2022-07-12T01:31:59.176395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf2 = RandomForestRegressor()\nrf2.fit(dataWindNot0_df,dataWindNot0_series)\npredicted = rf2.predict(dataWind0_df)\nprint(predicted)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:59.178654Z","iopub.execute_input":"2022-07-12T01:31:59.179197Z","iopub.status.idle":"2022-07-12T01:32:04.675437Z","shell.execute_reply.started":"2022-07-12T01:31:59.179166Z","shell.execute_reply":"2022-07-12T01:32:04.674365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataWind0['windspeed'] = predicted","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:32:04.676796Z","iopub.execute_input":"2022-07-12T01:32:04.677141Z","iopub.status.idle":"2022-07-12T01:32:04.682598Z","shell.execute_reply.started":"2022-07-12T01:32:04.677111Z","shell.execute_reply":"2022-07-12T01:32:04.681463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_dt=pd.concat([dataWind0,dataWindNot0],axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:32:04.684072Z","iopub.execute_input":"2022-07-12T01:32:04.685063Z","iopub.status.idle":"2022-07-12T01:32:04.702304Z","shell.execute_reply.started":"2022-07-12T01:32:04.685030Z","shell.execute_reply":"2022-07-12T01:32:04.701226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 컬럼 중 값이 일정하면 카테고리로 변경 필요하지 않은 컬럼들은 제거함.\ncategorizational_columns = ['holiday','humidity','season','weather','workingday','year','month','day','hour']\ndrop_columns = ['datetime','casual','registered','count','tempDate']","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:32:04.703828Z","iopub.execute_input":"2022-07-12T01:32:04.704450Z","iopub.status.idle":"2022-07-12T01:32:04.709220Z","shell.execute_reply.started":"2022-07-12T01:32:04.704418Z","shell.execute_reply":"2022-07-12T01:32:04.708335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# all_dt 카테고리로 변환\nfor col in categorizational_columns:\n    all_dt[col] = all_dt[col].astype('category')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:32:04.710669Z","iopub.execute_input":"2022-07-12T01:32:04.711336Z","iopub.status.idle":"2022-07-12T01:32:04.732467Z","shell.execute_reply.started":"2022-07-12T01:32:04.711305Z","shell.execute_reply":"2022-07-12T01:32:04.731625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 데이터에서 count가 있고 없고에 따 train, test를 분리\ntrain = all_dt[pd.notnull(all_dt['count'])].sort_values(by='datetime')\ntest = all_dt[~pd.notnull(all_dt['count'])].sort_values(by='datetime')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:32:04.738012Z","iopub.execute_input":"2022-07-12T01:32:04.738669Z","iopub.status.idle":"2022-07-12T01:32:04.774217Z","shell.execute_reply.started":"2022-07-12T01:32:04.738634Z","shell.execute_reply":"2022-07-12T01:32:04.773218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#데이터 훈련시 집어 넣게 될 각각의 결과 값들\ndatetimecol = test['datetime']\nyLabels = train['count']\nyLabelsRegistered = train['registered']# 등록된 사용자\nyLabelsCasual = train['casual'] # 임시 사용","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:32:04.775674Z","iopub.execute_input":"2022-07-12T01:32:04.776258Z","iopub.status.idle":"2022-07-12T01:32:04.781435Z","shell.execute_reply.started":"2022-07-12T01:32:04.776226Z","shell.execute_reply":"2022-07-12T01:32:04.780606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train, test에서 필요없는 columns을 제거\ntrain = train.drop(drop_columns,axis = 1)\ntest = test.drop(drop_columns,axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:32:04.782813Z","iopub.execute_input":"2022-07-12T01:32:04.783120Z","iopub.status.idle":"2022-07-12T01:32:04.795553Z","shell.execute_reply.started":"2022-07-12T01:32:04.783092Z","shell.execute_reply":"2022-07-12T01:32:04.794639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\"\"\"\nRMSLE를 활용하여 제대로 예측이 되었는지 평가.\n\nRMSLE\n과대평가 된 항목보다는 과소평가 된 항목을 확인\n오차를 제곱하여 평균한 값의 제곱근으로 값이 작아질 수록 정밀도가 높음\n0에 가까운 값이 나올 수록 정밀도가 높\n\"\"\"","metadata":{}},{"cell_type":"code","source":"# y is predict value y_ is actual value\ndef rmsle(y, y_,convertExp=True):\n    if convertExp:\n        y = np.exp(y), \n        y_ = np.exp(y_)\n    log1 = np.nan_to_num(np.array([np.log(v + 1) for v in y]))\n    log2 = np.nan_to_num(np.array([np.log(v + 1) for v in y_]))\n    calc = (log1 - log2) ** 2\n    return np.sqrt(np.mean(calc))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:32:04.797168Z","iopub.execute_input":"2022-07-12T01:32:04.798089Z","iopub.status.idle":"2022-07-12T01:32:04.806021Z","shell.execute_reply.started":"2022-07-12T01:32:04.798057Z","shell.execute_reply":"2022-07-12T01:32:04.805078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- 선형 회귀 모델","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression,Ridge,Lasso\nlr = LinearRegression()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:32:04.807172Z","iopub.execute_input":"2022-07-12T01:32:04.807626Z","iopub.status.idle":"2022-07-12T01:32:04.818473Z","shell.execute_reply.started":"2022-07-12T01:32:04.807598Z","shell.execute_reply":"2022-07-12T01:32:04.817416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\"\"\"\nnp.log1p는 np.log(1+x)와 동일. 어떤 x값이 0 일 때 log하게되면, (-)무한대를 수렴하기 때문에 np.log1p를 활용함. \n\"\"\"","metadata":{}},{"cell_type":"code","source":"yLabelslog = np.log1p(yLabels)\n# 선형 모델에 데이터 학습\nlr.fit(train,yLabelslog)\n# 결과\npreds = lr.predict(train)\n# 로그 값이 아닌, 원래 모델의 값을 넣기 위함\nprint('RMSLE Value For Linear Regression: {}'.format(rmsle(np.exp(yLabelslog),np.exp(preds),False)))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:32:04.819825Z","iopub.execute_input":"2022-07-12T01:32:04.820134Z","iopub.status.idle":"2022-07-12T01:32:05.028659Z","shell.execute_reply.started":"2022-07-12T01:32:04.820099Z","shell.execute_reply":"2022-07-12T01:32:05.027641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# count 값의 분포도를 distplot을 활용하여 확인\nsns.distplot(yLabels,bins=range(yLabels.min().astype('int'),yLabels.max().astype('int')))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:32:05.030109Z","iopub.execute_input":"2022-07-12T01:32:05.030526Z","iopub.status.idle":"2022-07-12T01:32:07.470050Z","shell.execute_reply.started":"2022-07-12T01:32:05.030484Z","shell.execute_reply":"2022-07-12T01:32:07.469291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(yLabels.count())","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:32:07.471227Z","iopub.execute_input":"2022-07-12T01:32:07.471960Z","iopub.status.idle":"2022-07-12T01:32:07.476922Z","shell.execute_reply.started":"2022-07-12T01:32:07.471924Z","shell.execute_reply":"2022-07-12T01:32:07.475786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 이상치 배제 train count 개수\nyLabels[np.logical_and(yLabels.mean()-3*yLabels.std() <= yLabels,yLabels.mean()+3*yLabels.std() >= yLabels)].count()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:32:07.478458Z","iopub.execute_input":"2022-07-12T01:32:07.478982Z","iopub.status.idle":"2022-07-12T01:32:07.494093Z","shell.execute_reply.started":"2022-07-12T01:32:07.478928Z","shell.execute_reply":"2022-07-12T01:32:07.492904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# GridSearchCV를 활용하여 파라미터 튜닝시 어떤 파라미터가 최적의 값을 내는지 등을 확\n\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn import metrics","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:32:07.495827Z","iopub.execute_input":"2022-07-12T01:32:07.496138Z","iopub.status.idle":"2022-07-12T01:32:07.504517Z","shell.execute_reply.started":"2022-07-12T01:32:07.496110Z","shell.execute_reply":"2022-07-12T01:32:07.503415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ridge의 파라미터 중 어떤 파라미터에 배열 값으로 넘겨주면 최적의 값을 알려줌\nridge = Ridge()\nridge_params = {'max_iter':[3000],'alpha':[0.001,0.01,0.1,1,10,100,1000]}\nrmsle_scorer = metrics.make_scorer(rmsle,greater_is_better=False)\ngrid_ridge = GridSearchCV(ridge,ridge_params,scoring=rmsle_scorer,cv=5)\n\ngrid_ridge.fit(train,yLabelslog)\npreds = grid_ridge.predict(train)\nprint(grid_ridge.best_params_)\nprint('RMSLE Value for Ridge Regression {}'.format(rmsle(np.exp(yLabelslog),np.exp(preds),False)))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:32:07.505846Z","iopub.execute_input":"2022-07-12T01:32:07.506341Z","iopub.status.idle":"2022-07-12T01:32:09.734631Z","shell.execute_reply.started":"2022-07-12T01:32:07.506303Z","shell.execute_reply":"2022-07-12T01:32:09.733483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cv result를 통해 alpha값의 변화에 따른 평균 변화 파악\ndf = pd.DataFrame(grid_ridge.cv_results_)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:32:09.736126Z","iopub.execute_input":"2022-07-12T01:32:09.736807Z","iopub.status.idle":"2022-07-12T01:32:09.744913Z","shell.execute_reply.started":"2022-07-12T01:32:09.736764Z","shell.execute_reply":"2022-07-12T01:32:09.743735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:32:09.746946Z","iopub.execute_input":"2022-07-12T01:32:09.748089Z","iopub.status.idle":"2022-07-12T01:32:09.776081Z","shell.execute_reply.started":"2022-07-12T01:32:09.748038Z","shell.execute_reply":"2022-07-12T01:32:09.775241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lasso = Lasso()\n\nlasso_params = {'max_iter':[3000],'alpha':[0.001,0.01,0.1,1,10,100,1000]}\ngrid_lasso = GridSearchCV(lasso,lasso_params,scoring=rmsle_scorer,cv=5)\ngrid_lasso.fit(train,yLabelslog)\npreds = grid_lasso.predict(train)\nprint('RMSLE Value for Lasso Regression {}'.format(rmsle(np.exp(yLabelslog),np.exp(preds),False)))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:42:33.624598Z","iopub.execute_input":"2022-07-12T01:42:33.625002Z","iopub.status.idle":"2022-07-12T01:42:37.218553Z","shell.execute_reply.started":"2022-07-12T01:42:33.624969Z","shell.execute_reply":"2022-07-12T01:42:37.217357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf = RandomForestRegressor()\n\nrf_params = {'n_estimators':[1,10,100]}\ngrid_rf = GridSearchCV(rf,rf_params,scoring=rmsle_scorer,cv=5)\ngrid_rf.fit(train,yLabelslog)\npreds = grid_rf.predict(train)\nprint('RMSLE Value for RandomForest {}'.format(rmsle(np.exp(yLabelslog),np.exp(preds),False)))\n\nfrom sklearn.ensemble import GradientBoostingRegressor\n\ngb = GradientBoostingRegressor()\ngb_params={'max_depth':range(1,11,1),'n_estimators':[1,10,100]}\ngrid_gb=GridSearchCV(gb,gb_params,scoring=rmsle_scorer,cv=5)\ngrid_gb.fit(train,yLabelslog)\npreds = grid_gb.predict(train)\nprint('RMSLE Value for GradientBoosting {}'.format(rmsle(np.exp(yLabelslog),np.exp(preds),False)))\n\n\npredsTest = grid_gb.predict(test)\nfig,(ax1,ax2)= plt.subplots(ncols=2)\nfig.set_size_inches(12,5)\nsns.distplot(yLabels,ax=ax1,bins=50)\nsns.distplot(np.exp(predsTest),ax=ax2,bins=50)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:42:44.836265Z","iopub.execute_input":"2022-07-12T01:42:44.836646Z","iopub.status.idle":"2022-07-12T01:44:43.462740Z","shell.execute_reply.started":"2022-07-12T01:42:44.836614Z","shell.execute_reply":"2022-07-12T01:44:43.461644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\n        \"datetime\": datetimecol,\n        \"count\": [max(0, x) for x in np.exp(predsTest)]\n    })\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:45:31.389839Z","iopub.execute_input":"2022-07-12T01:45:31.390261Z","iopub.status.idle":"2022-07-12T01:45:31.439493Z","shell.execute_reply.started":"2022-07-12T01:45:31.390227Z","shell.execute_reply":"2022-07-12T01:45:31.438644Z"},"trusted":true},"execution_count":null,"outputs":[]}]}