{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n    \ndata_path = '/kaggle/input/bike-sharing-demand/'    \n\ntrain = pd.read_csv(data_path +'train.csv')\ntest = pd.read_csv(data_path + 'test.csv')\nsubmission = pd.read_csv(data_path + 'sampleSubmission.csv')\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-28T13:08:09.371123Z","iopub.execute_input":"2022-07-28T13:08:09.371537Z","iopub.status.idle":"2022-07-28T13:08:09.426690Z","shell.execute_reply.started":"2022-07-28T13:08:09.371504Z","shell.execute_reply":"2022-07-28T13:08:09.425882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 이상치 제거 (폭우 폭설날 대여 건)\ntrain = train[train['weather'] != 4]\n\n# 데이터 합치기\nall_data = pd.concat([train, test], ignore_index=True)\nall_data","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:08:09.428156Z","iopub.execute_input":"2022-07-28T13:08:09.428618Z","iopub.status.idle":"2022-07-28T13:08:09.458138Z","shell.execute_reply.started":"2022-07-28T13:08:09.428590Z","shell.execute_reply":"2022-07-28T13:08:09.457175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 파생 피쳐 추가","metadata":{}},{"cell_type":"code","source":"from datetime import datetime\n\nall_data['date'] = all_data['datetime'].apply(lambda x : x.split()[0]) #찐 날짜를 하려면 년월일이 나와야 함\nall_data['year'] = all_data['datetime'].apply(lambda x : x.split()[0].split('-')[0]) #연도별\nall_data['month'] = all_data['datetime'].apply(lambda x : x.split()[0].split('-')[1]) #월별\nall_data['hour'] = all_data['datetime'].apply(lambda x : x.split()[1].split(':')[0]) #시간대별\nall_data[\"weekday\"] = all_data['date'].apply(lambda dateString : datetime.strptime(dateString, \"%Y-%m-%d\").weekday()) # 요일","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:08:09.464608Z","iopub.execute_input":"2022-07-28T13:08:09.465260Z","iopub.status.idle":"2022-07-28T13:08:09.750554Z","shell.execute_reply.started":"2022-07-28T13:08:09.465226Z","shell.execute_reply":"2022-07-28T13:08:09.749664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drop_features = ['casual', 'registered', 'datetime', 'date', 'windspeed', 'month'] # 불필요한 피쳐 제거\nall_data = all_data.drop(drop_features, axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:08:09.752109Z","iopub.execute_input":"2022-07-28T13:08:09.752587Z","iopub.status.idle":"2022-07-28T13:08:09.763926Z","shell.execute_reply.started":"2022-07-28T13:08:09.752560Z","shell.execute_reply":"2022-07-28T13:08:09.762949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = all_data[~pd.isnull(all_data['count'])]\nX_test = all_data[pd.isnull(all_data['count'])]\n\nX_train = X_train.drop(['count'], axis = 1)\nX_test = X_test.drop(['count'], axis = 1)\n\ny = train['count']","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:08:09.765171Z","iopub.execute_input":"2022-07-28T13:08:09.765737Z","iopub.status.idle":"2022-07-28T13:08:09.776762Z","shell.execute_reply.started":"2022-07-28T13:08:09.765707Z","shell.execute_reply":"2022-07-28T13:08:09.775520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:08:09.779079Z","iopub.execute_input":"2022-07-28T13:08:09.779458Z","iopub.status.idle":"2022-07-28T13:08:09.795673Z","shell.execute_reply.started":"2022-07-28T13:08:09.779423Z","shell.execute_reply":"2022-07-28T13:08:09.794578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 평가지표 개산 함수 작성 ( rmsle라는 함수(클래스) 생성)","metadata":{}},{"cell_type":"code","source":"import numpy as np\ndef rmsle(y_true, y_pred, convertExp = True): #y_true는 실제 타깃값, y_pred는 예측값\n    # 지수변환 함수 = convertExp, exp()\n    # log(count)를 타깃값으로 사용하기 때문에 미리 지수변환 \n    # 즉, rmsle에 던져지는 값들이 앞으로의 계산에 바로 사용될 수 있는 상태가 아니라 정규분포 만들려고 로그취해진 값임\n    if convertExp:\n        y_true = np.exp(y_true)\n        y_pred = np.exp(y_pred)\n\n    # RMSLE 계산용 로그변환(0일 때 -무한대로 갈 수 있으니까 +1인 채로 로그변환) 후 결측값을 0으로 변환(np.nan_to_num)\n    log_true = np.nan_to_num(np.log(y_true+1))\n    log_pred = np.nan_to_num(np.log(y_pred+1))\n    \n    # RMSLE 계산\n    output = np.sqrt(np.mean((log_true - log_pred)**2)) # 로그변환은 그냥 하는 것이 아니라 +1해서 한다고 위에서 정의\n    return output","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:08:09.797141Z","iopub.execute_input":"2022-07-28T13:08:09.797700Z","iopub.status.idle":"2022-07-28T13:08:09.805195Z","shell.execute_reply.started":"2022-07-28T13:08:09.797669Z","shell.execute_reply":"2022-07-28T13:08:09.804130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 선형회귀모델 불러옴","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\n\nlinear_reg_model = LinearRegression() #선형회귀모델","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:08:09.806759Z","iopub.execute_input":"2022-07-28T13:08:09.807725Z","iopub.status.idle":"2022-07-28T13:08:09.817015Z","shell.execute_reply.started":"2022-07-28T13:08:09.807686Z","shell.execute_reply":"2022-07-28T13:08:09.815750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 로그변환해서 정규분포곡선과 유사하게 한 후 모델 훈련시키기","metadata":{}},{"cell_type":"code","source":"log_y = np.log(y) #타깃값 로그변환\nlinear_reg_model.fit(X_train, log_y) # 모델 훈련(선형회귀모델)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:08:09.818776Z","iopub.execute_input":"2022-07-28T13:08:09.819649Z","iopub.status.idle":"2022-07-28T13:08:09.860871Z","shell.execute_reply.started":"2022-07-28T13:08:09.819568Z","shell.execute_reply":"2022-07-28T13:08:09.859260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"탐색적 데이터 분석: 예측에 도움이 될 피쳐를 추리고, 적절한 모델링 방법을 탐색하는 과정\n\n피처 엔지니어링: 추려진 피처들을 훈련에 적합하도록, 성능 향상에 도움되도록 가공하는 과정","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"preds = linear_reg_model.predict(X_train) # 만든 모델로 다시 훈련 데이터를 예측하여 preds에 넣기. 원래는 훈련데이터로 예측X\nprint(f'선형회귀의 RMSLE 값 : {rmsle(log_y, preds, True):.4f}') #예측값(preds)과 실제 값(log_y) 비교하여 선형회귀 RMSLE 값\n# 이 때, rmsle()에 들어가는 y에만 로그값인 이유는 우리가 구한 preds는 log_y를 이용해 훈련시킨 모델로 예측한 값이므로 이미 log가 씌워진 상태ㅕ서","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:08:09.862852Z","iopub.execute_input":"2022-07-28T13:08:09.864057Z","iopub.status.idle":"2022-07-28T13:08:09.915188Z","shell.execute_reply.started":"2022-07-28T13:08:09.863989Z","shell.execute_reply":"2022-07-28T13:08:09.913690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 테스트데이터로 예측 및 결과 제출","metadata":{}},{"cell_type":"code","source":"linearreg_preds = linear_reg_model.predict(X_test) #테스트 데이터로 예측\n\nsubmission['count'] = np.exp(linearreg_preds) #지수변환\nsubmission.to_csv('submission.csv', index = False) #파일로 저장","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:08:09.917271Z","iopub.execute_input":"2022-07-28T13:08:09.917740Z","iopub.status.idle":"2022-07-28T13:08:10.021096Z","shell.execute_reply.started":"2022-07-28T13:08:09.917699Z","shell.execute_reply":"2022-07-28T13:08:10.019604Z"},"trusted":true},"execution_count":null,"outputs":[]}]}