{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# 1. import packages\n\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.svm import SVC\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.datasets import make_classification\nfrom sklearn.model_selection import train_test_split\n\nimport pandas as pd\nimport numpy as np\nfrom scipy import stats\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LinearRegression, Ridge, Lasso, ElasticNet\nfrom sklearn.linear_model import LassoCV , ElasticNetCV , RidgeCV\nfrom sklearn.preprocessing import PolynomialFeatures\nfrom sklearn.decomposition import PCA \nfrom sklearn.metrics import mean_squared_error as mse\nfrom sklearn.metrics import mean_squared_log_error as msle \nfrom sklearn.metrics import r2_score as r2\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.ensemble import GradientBoostingRegressor as GBR\nfrom sklearn.ensemble import RandomForestRegressor as RFR\nfrom sklearn.cross_decomposition import PLSRegression as  PLS\nfrom sklearn.svm import SVR\nfrom sklearn.kernel_ridge import KernelRidge\nfrom sklearn.model_selection import ShuffleSplit\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import GridSearchCV","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:07:48.342576Z","iopub.execute_input":"2022-07-15T00:07:48.342916Z","iopub.status.idle":"2022-07-15T00:07:48.352406Z","shell.execute_reply.started":"2022-07-15T00:07:48.342892Z","shell.execute_reply":"2022-07-15T00:07:48.351059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 2. load datasets\n\ntrain = pd.read_csv(\"../input/bike-sharing-demand/train.csv\")\ntest = pd.read_csv(\"../input/bike-sharing-demand/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:07:48.354858Z","iopub.execute_input":"2022-07-15T00:07:48.355849Z","iopub.status.idle":"2022-07-15T00:07:48.394198Z","shell.execute_reply.started":"2022-07-15T00:07:48.355802Z","shell.execute_reply":"2022-07-15T00:07:48.392823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.tail()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:07:48.395925Z","iopub.execute_input":"2022-07-15T00:07:48.396183Z","iopub.status.idle":"2022-07-15T00:07:48.408514Z","shell.execute_reply.started":"2022-07-15T00:07:48.396160Z","shell.execute_reply":"2022-07-15T00:07:48.407810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.tail()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:07:48.409427Z","iopub.execute_input":"2022-07-15T00:07:48.409901Z","iopub.status.idle":"2022-07-15T00:07:48.427366Z","shell.execute_reply.started":"2022-07-15T00:07:48.409875Z","shell.execute_reply":"2022-07-15T00:07:48.426520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 3. preprecessing\n\ntrain['datetime'] = pd.to_datetime(train['datetime'])\ntrain['hour'] = train['datetime'].dt.hour\ntrain['month'] = train['datetime'].dt.month\ntrain[\"dayofweek\"] = train['datetime'].dt.dayofweek\ntrain[\"year\"] = train['datetime'].dt.year\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:07:48.429403Z","iopub.execute_input":"2022-07-15T00:07:48.429653Z","iopub.status.idle":"2022-07-15T00:07:48.444582Z","shell.execute_reply.started":"2022-07-15T00:07:48.429629Z","shell.execute_reply":"2022-07-15T00:07:48.443832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## 3.1 create new variables (via reshape (ffill) & rolling)\n\nprint(train.shape) \ntrain_ffill = train.set_index('datetime').resample('H').ffill()\ntrain_ffill = train_ffill.reset_index()\nprint(train_ffill.shape) ## 누락된 시간 6,370 개 (이전 데이터로 ffill)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:07:48.445550Z","iopub.execute_input":"2022-07-15T00:07:48.445943Z","iopub.status.idle":"2022-07-15T00:07:48.457661Z","shell.execute_reply.started":"2022-07-15T00:07:48.445919Z","shell.execute_reply":"2022-07-15T00:07:48.456637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = pd.merge(train_ffill, train, how ='left', on = 'datetime')\ntemp[temp.month_y.isna()].datetime","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:07:48.458652Z","iopub.execute_input":"2022-07-15T00:07:48.459357Z","iopub.status.idle":"2022-07-15T00:07:48.481763Z","shell.execute_reply.started":"2022-07-15T00:07:48.459332Z","shell.execute_reply":"2022-07-15T00:07:48.480941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## 1) 1시간 이전 데이터\ntrain_ffill[['temp_prev', 'atemp_prev', 'humidity_prev', 'windspeed_prev']] = train_ffill[['temp', 'atemp', 'humidity', 'windspeed']].shift(1)\ntrain_ffill = train_ffill.copy()\ntrain_ffill.loc[0,['temp_prev', 'atemp_prev', 'humidity_prev', 'windspeed_prev']] = train_ffill[['temp', 'atemp', 'humidity', 'windspeed']].iloc[0].values\n\n## 2) 이전 4시간동안 평균 \ntrain_ffill_4h = train_ffill[['temp_prev', 'atemp_prev', 'humidity_prev', 'windspeed_prev']].rolling(window=4).mean()\ntrain_ffill_4h.loc[0:2,['temp_prev', 'atemp_prev', 'humidity_prev', 'windspeed_prev']] = train_ffill.loc[0:2,['temp', 'atemp', 'humidity', 'windspeed']].values\ntrain_ffill_4h.columns = ['temp_prev_4h', 'atemp_prev_4h', 'humidity_prev_4h', 'windspeed_prev_4h']\n\n## 3) 이전 12시간동안 평균\ntrain_ffill_12h = train_ffill[['temp_prev', 'atemp_prev', 'humidity_prev', 'windspeed_prev']].rolling(window=12).mean()\ntrain_ffill_12h.loc[0:10,['temp_prev', 'atemp_prev', 'humidity_prev', 'windspeed_prev']] = train_ffill.loc[0:10,['temp', 'atemp', 'humidity', 'windspeed']].values\ntrain_ffill_12h.columns = ['temp_prev_12h', 'atemp_prev_12h', 'humidity_prev_12h', 'windspeed_prev_12h']\n\n## 4) 이전 24시간동안 평균\ntrain_ffill_24h = train_ffill[['temp_prev', 'atemp_prev', 'humidity_prev', 'windspeed_prev']].rolling(window=24).mean()\ntrain_ffill_24h.loc[0:22,['temp_prev', 'atemp_prev', 'humidity_prev', 'windspeed_prev']] = train_ffill.loc[0:22,['temp', 'atemp', 'humidity', 'windspeed']].values\ntrain_ffill_24h.columns = ['temp_prev_24h', 'atemp_prev_24h', 'humidity_prev_24h', 'windspeed_prev_24h']\n\n\ntrain_ffill = pd.concat([train_ffill, train_ffill_4h], axis = 1)\ntrain_ffill = pd.concat([train_ffill, train_ffill_12h], axis = 1)\ntrain_ffill = pd.concat([train_ffill, train_ffill_24h], axis = 1)\ntrain_ffill.head()\n\ntrain = pd.merge(train_ffill, train[['datetime']], how ='right', on = 'datetime')","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:07:48.483230Z","iopub.execute_input":"2022-07-15T00:07:48.483465Z","iopub.status.idle":"2022-07-15T00:07:48.523724Z","shell.execute_reply.started":"2022-07-15T00:07:48.483443Z","shell.execute_reply":"2022-07-15T00:07:48.522646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.tail(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:07:48.594539Z","iopub.execute_input":"2022-07-15T00:07:48.594881Z","iopub.status.idle":"2022-07-15T00:07:48.618269Z","shell.execute_reply.started":"2022-07-15T00:07:48.594856Z","shell.execute_reply":"2022-07-15T00:07:48.617646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:07:48.620130Z","iopub.execute_input":"2022-07-15T00:07:48.621165Z","iopub.status.idle":"2022-07-15T00:07:48.628682Z","shell.execute_reply.started":"2022-07-15T00:07:48.621131Z","shell.execute_reply":"2022-07-15T00:07:48.627576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test set에도 동일하게 new variable 생성\n\ntest['datetime'] = pd.to_datetime(test['datetime'])\ntest['hour'] = test['datetime'].dt.hour\ntest['month'] = test['datetime'].dt.month\ntest[\"dayofweek\"] = test['datetime'].dt.dayofweek\ntest[\"year\"] = test['datetime'].dt.year\n\nprint(test.shape) \ntest_ffill = test.set_index('datetime').resample('H').ffill()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:07:48.629885Z","iopub.execute_input":"2022-07-15T00:07:48.631061Z","iopub.status.idle":"2022-07-15T00:07:48.652056Z","shell.execute_reply.started":"2022-07-15T00:07:48.630999Z","shell.execute_reply":"2022-07-15T00:07:48.650986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ffill = test_ffill.reset_index()\nprint(test_ffill.shape) ## 누락된 시간 10,595 개","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:07:48.654232Z","iopub.execute_input":"2022-07-15T00:07:48.654797Z","iopub.status.idle":"2022-07-15T00:07:48.660958Z","shell.execute_reply.started":"2022-07-15T00:07:48.654768Z","shell.execute_reply":"2022-07-15T00:07:48.659904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## 1) 1시간 이전 데이터\ntest_ffill[['temp_prev', 'atemp_prev', 'humidity_prev', 'windspeed_prev']] = test_ffill[['temp', 'atemp', 'humidity', 'windspeed']].shift(1)\ntest_ffill = test_ffill.copy()\ntest_ffill.loc[0,['temp_prev', 'atemp_prev', 'humidity_prev', 'windspeed_prev']] = test_ffill[['temp', 'atemp', 'humidity', 'windspeed']].iloc[0].values\n\n## 2) 이전 4시간동안 평균 \ntest_ffill_4h = test_ffill[['temp_prev', 'atemp_prev', 'humidity_prev', 'windspeed_prev']].rolling(window=4).mean()\ntest_ffill_4h.loc[0:2,['temp_prev', 'atemp_prev', 'humidity_prev', 'windspeed_prev']] = test_ffill.loc[0:2,['temp', 'atemp', 'humidity', 'windspeed']].values\ntest_ffill_4h.columns = ['temp_prev_4h', 'atemp_prev_4h', 'humidity_prev_4h', 'windspeed_prev_4h']\n\n## 3) 이전 12시간동안 평균\ntest_ffill_12h = test_ffill[['temp_prev', 'atemp_prev', 'humidity_prev', 'windspeed_prev']].rolling(window=12).mean()\ntest_ffill_12h.loc[0:10,['temp_prev', 'atemp_prev', 'humidity_prev', 'windspeed_prev']] = test_ffill.loc[0:10,['temp', 'atemp', 'humidity', 'windspeed']].values\ntest_ffill_12h.columns = ['temp_prev_12h', 'atemp_prev_12h', 'humidity_prev_12h', 'windspeed_prev_12h']\n\n## 4) 이전 24시간동안 평균\ntest_ffill_24h = test_ffill[['temp_prev', 'atemp_prev', 'humidity_prev', 'windspeed_prev']].rolling(window=24).mean()\ntest_ffill_24h.loc[0:22,['temp_prev', 'atemp_prev', 'humidity_prev', 'windspeed_prev']] = test_ffill.loc[0:22,['temp', 'atemp', 'humidity', 'windspeed']].values\ntest_ffill_24h.columns = ['temp_prev_24h', 'atemp_prev_24h', 'humidity_prev_24h', 'windspeed_prev_24h']\n\ntest_ffill = pd.concat([test_ffill, test_ffill_4h], axis = 1)\ntest_ffill = pd.concat([test_ffill, test_ffill_12h], axis = 1)\ntest_ffill = pd.concat([test_ffill, test_ffill_24h], axis = 1)\ntest_ffill.head()\n\ntest = pd.merge(test_ffill, test[['datetime']], how ='right', on = 'datetime')","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:07:48.664241Z","iopub.execute_input":"2022-07-15T00:07:48.664568Z","iopub.status.idle":"2022-07-15T00:07:48.715259Z","shell.execute_reply.started":"2022-07-15T00:07:48.664541Z","shell.execute_reply":"2022-07-15T00:07:48.712734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.tail()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:07:48.719455Z","iopub.execute_input":"2022-07-15T00:07:48.722306Z","iopub.status.idle":"2022-07-15T00:07:48.748957Z","shell.execute_reply.started":"2022-07-15T00:07:48.722263Z","shell.execute_reply":"2022-07-15T00:07:48.747610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create dummy variables\n\ntrain_X = train[['season', 'holiday', 'workingday', 'weather', 'temp','atemp', 'humidity', 'windspeed', 'hour', 'month', 'dayofweek', 'year', 'temp_prev', 'atemp_prev','humidity_prev', 'windspeed_prev', 'temp_prev_4h', 'atemp_prev_4h','humidity_prev_4h', 'windspeed_prev_4h', 'temp_prev_12h', 'atemp_prev_12h', 'humidity_prev_12h', 'windspeed_prev_12h','temp_prev_24h', 'atemp_prev_24h', 'humidity_prev_24h', 'windspeed_prev_24h', 'temp_prev_24h', 'atemp_prev_24h','humidity_prev_24h', 'windspeed_prev_24h']]\ntrain_X = pd.get_dummies(data=train_X, columns=['season','weather','dayofweek'],drop_first=True)\ntrain_Y = train[['count']]\n\n\ntest_X = test[['season', 'holiday', 'workingday', 'weather', 'temp','atemp', 'humidity', 'windspeed', 'hour', 'month', 'dayofweek', 'year', 'temp_prev', 'atemp_prev','humidity_prev', 'windspeed_prev', 'temp_prev_4h', 'atemp_prev_4h','humidity_prev_4h', 'windspeed_prev_4h', 'temp_prev_12h', 'atemp_prev_12h', 'humidity_prev_12h', 'windspeed_prev_12h','temp_prev_24h', 'atemp_prev_24h', 'humidity_prev_24h', 'windspeed_prev_24h', 'temp_prev_24h', 'atemp_prev_24h','humidity_prev_24h', 'windspeed_prev_24h']]\ntest_X = pd.get_dummies(data=test_X, columns=['season','weather','dayofweek'],drop_first=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:07:48.751143Z","iopub.execute_input":"2022-07-15T00:07:48.752334Z","iopub.status.idle":"2022-07-15T00:07:48.783747Z","shell.execute_reply.started":"2022-07-15T00:07:48.752300Z","shell.execute_reply":"2022-07-15T00:07:48.782579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:07:48.785166Z","iopub.execute_input":"2022-07-15T00:07:48.785493Z","iopub.status.idle":"2022-07-15T00:07:48.793119Z","shell.execute_reply.started":"2022-07-15T00:07:48.785465Z","shell.execute_reply":"2022-07-15T00:07:48.791994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_X.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:07:48.794686Z","iopub.execute_input":"2022-07-15T00:07:48.795250Z","iopub.status.idle":"2022-07-15T00:07:48.806238Z","shell.execute_reply.started":"2022-07-15T00:07:48.795214Z","shell.execute_reply":"2022-07-15T00:07:48.805289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 4. train/val split (20%)\n\ntrain_x, val_x, train_y, val_y = train_test_split(train_X, train_Y, test_size = 0.2, random_state = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:07:48.807446Z","iopub.execute_input":"2022-07-15T00:07:48.808134Z","iopub.status.idle":"2022-07-15T00:07:48.825628Z","shell.execute_reply.started":"2022-07-15T00:07:48.808105Z","shell.execute_reply":"2022-07-15T00:07:48.824546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 5. train\n\n# 5.1 build pipline\nfrom sklearn.ensemble import GradientBoostingRegressor, RandomForestRegressor\nfrom xgboost import XGBRegressor\npipelines = []\n\n# pipelines.append(('PolyScaledGBM', Pipeline([('poly', PolynomialFeatures()), ('Scaler', StandardScaler()), ('GBM', GradientBoostingRegressor(random_state=42))])))\npipelines.append(('PolyScaledRFR', Pipeline([('poly', PolynomialFeatures()), ('Scaler', StandardScaler()), ('RFR', RandomForestRegressor(random_state=42))])))\n# pipelines.append(('PolyScaledSVR', Pipeline([('poly', PolynomialFeatures()), ('Scaler', StandardScaler()), ('SVR', SVR(kernel='linear'))])))\npipelines.append(('PolyScaledXGBR', Pipeline([('poly', PolynomialFeatures()), ('Scaler', StandardScaler()), ('XGBR', XGBRegressor(random_state=42))])))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:07:48.827338Z","iopub.execute_input":"2022-07-15T00:07:48.827871Z","iopub.status.idle":"2022-07-15T00:07:48.834962Z","shell.execute_reply.started":"2022-07-15T00:07:48.827835Z","shell.execute_reply":"2022-07-15T00:07:48.834019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_x.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:07:48.836937Z","iopub.execute_input":"2022-07-15T00:07:48.838703Z","iopub.status.idle":"2022-07-15T00:07:48.854177Z","shell.execute_reply.started":"2022-07-15T00:07:48.838666Z","shell.execute_reply":"2022-07-15T00:07:48.853059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 5.2 train\n\n## original\n\ntrain_x1 = train_x[['holiday', 'workingday', 'temp', 'atemp', 'humidity', 'windspeed', 'hour', 'month', 'year','season_2', 'season_3', 'season_4', 'weather_2', 'weather_3', 'weather_4','dayofweek_1', 'dayofweek_2', 'dayofweek_3', 'dayofweek_4','dayofweek_5', 'dayofweek_6']]\n\nresults = []\nnames = []\nfor name, model in pipelines:\n    cv_results = -cross_val_score(model, train_x1, np.log(train_y[\"count\"]+1), cv=3, scoring='neg_mean_squared_error')\n    results.append(np.sqrt(cv_results))\n    names.append(name)\n    msg = \"{}: {} ({})\".format(name, cv_results.mean(), cv_results.std())\n    print(msg)\n    \n# BEST : PolyScaledXGBR: 0.09491486590059565 (0.006457416609639065)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:07:48.855338Z","iopub.execute_input":"2022-07-15T00:07:48.855581Z","iopub.status.idle":"2022-07-15T00:08:55.912545Z","shell.execute_reply.started":"2022-07-15T00:07:48.855558Z","shell.execute_reply":"2022-07-15T00:08:55.911840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## original + 1hour before  \n\ntrain_x1 = train_x[['holiday', 'workingday', 'temp', 'atemp', 'humidity', 'windspeed',  'hour', 'month', 'year','season_2', 'season_3', 'season_4', 'weather_2', 'weather_3', 'weather_4','dayofweek_1', 'dayofweek_2', 'dayofweek_3', 'dayofweek_4','dayofweek_5', 'dayofweek_6','temp_prev', 'atemp_prev', 'humidity_prev', 'windspeed_prev']]\n\nresults = []\nnames = []\nfor name, model in pipelines:\n    cv_results = -cross_val_score(model, train_x1, np.log(train_y[\"count\"]+1), cv=3, scoring='neg_mean_squared_error')\n    results.append(np.sqrt(cv_results))\n    names.append(name)\n    msg = \"{}: {} ({})\".format(name, cv_results.mean(), cv_results.std())\n    print(msg)\n\n# BEST : PolyScaledXGBR: 0.09950802131402235 (0.0037402167439483046)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:08:55.917799Z","iopub.execute_input":"2022-07-15T00:08:55.918475Z","iopub.status.idle":"2022-07-15T00:11:00.767217Z","shell.execute_reply.started":"2022-07-15T00:08:55.918447Z","shell.execute_reply":"2022-07-15T00:11:00.766498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## original + 4hour before\n\ntrain_x1 = train_x[['holiday', 'workingday', 'temp', 'atemp', 'humidity', 'windspeed', 'hour', 'month', 'year','season_2', 'season_3', 'season_4', 'weather_2', 'weather_3', 'weather_4','dayofweek_1', 'dayofweek_2', 'dayofweek_3', 'dayofweek_4','dayofweek_5', 'dayofweek_6','temp_prev_4h','atemp_prev_4h', 'humidity_prev_4h', 'windspeed_prev_4h']]\n\nresults = []\nnames = []\nfor name, model in pipelines:\n    cv_results = -cross_val_score(model, train_x1, np.log(train_y[\"count\"]+1), cv=3, scoring='neg_mean_squared_error')\n    results.append(np.sqrt(cv_results))\n    names.append(name)\n    msg = \"{}: {} ({})\".format(name, cv_results.mean(), cv_results.std())\n    print(msg)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:11:00.770206Z","iopub.execute_input":"2022-07-15T00:11:00.772166Z","iopub.status.idle":"2022-07-15T00:13:24.781796Z","shell.execute_reply.started":"2022-07-15T00:11:00.772132Z","shell.execute_reply":"2022-07-15T00:13:24.781120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## original + 12hour before\n\ntrain_x1 = train_x[['holiday', 'workingday', 'temp', 'atemp', 'humidity', 'windspeed', 'hour', 'month', 'year','season_2', 'season_3', 'season_4', 'weather_2', 'weather_3', 'weather_4','dayofweek_1', 'dayofweek_2', 'dayofweek_3', 'dayofweek_4','dayofweek_5', 'dayofweek_6','temp_prev_12h','atemp_prev_12h', 'humidity_prev_12h', 'windspeed_prev_12h']]\n\nresults = []\nnames = []\nfor name, model in pipelines:\n    cv_results = -cross_val_score(model, train_x1, np.log(train_y[\"count\"]+1), cv=3, scoring='neg_mean_squared_error')\n    results.append(np.sqrt(cv_results))\n    names.append(name)\n    msg = \"{}: {} ({})\".format(name, cv_results.mean(), cv_results.std())\n    print(msg)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:13:24.783146Z","iopub.execute_input":"2022-07-15T00:13:24.783626Z","iopub.status.idle":"2022-07-15T00:15:56.125975Z","shell.execute_reply.started":"2022-07-15T00:13:24.783575Z","shell.execute_reply":"2022-07-15T00:15:56.125262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 6. final prediction","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:15:56.128946Z","iopub.execute_input":"2022-07-15T00:15:56.130897Z","iopub.status.idle":"2022-07-15T00:15:56.135056Z","shell.execute_reply.started":"2022-07-15T00:15:56.130865Z","shell.execute_reply":"2022-07-15T00:15:56.134469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 6.1 original (0.45521)\n\nX = train_X[['holiday', 'workingday', 'temp', 'atemp', 'humidity', 'windspeed', 'hour', 'month', 'year','season_2', 'season_3', 'season_4', 'weather_2', 'weather_3', 'weather_4','dayofweek_1', 'dayofweek_2', 'dayofweek_3', 'dayofweek_4','dayofweek_5', 'dayofweek_6']]\nX_t = test_X[['holiday', 'workingday', 'temp', 'atemp', 'humidity', 'windspeed', 'hour', 'month', 'year','season_2', 'season_3', 'season_4', 'weather_2', 'weather_3', 'weather_4','dayofweek_1', 'dayofweek_2', 'dayofweek_3', 'dayofweek_4','dayofweek_5', 'dayofweek_6']]\n\npipe = pipelines[1][1]\nprint(pipe)\npipe.fit(X, np.log(train_Y[\"count\"]+1))\ny_pred = np.exp(pipe.predict(X_t))-1\n\nsub = test[[\"datetime\"]]\nsub[\"count\"] = y_pred\nsub.to_csv(\"Submission_ori.csv\", index = False)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-07-15T00:15:56.136974Z","iopub.execute_input":"2022-07-15T00:15:56.137552Z","iopub.status.idle":"2022-07-15T00:16:00.920083Z","shell.execute_reply.started":"2022-07-15T00:15:56.137521Z","shell.execute_reply":"2022-07-15T00:16:00.918722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 6.2 1 hour before (0.44480)\n\nX = train_X[['holiday', 'workingday', 'temp', 'atemp', 'humidity', 'windspeed',  'hour', 'month', 'year','season_2', 'season_3', 'season_4', 'weather_2', 'weather_3', 'weather_4','dayofweek_1', 'dayofweek_2', 'dayofweek_3', 'dayofweek_4','dayofweek_5', 'dayofweek_6','temp_prev', 'atemp_prev', 'humidity_prev', 'windspeed_prev']]\nX_t = test_X[['holiday', 'workingday', 'temp', 'atemp', 'humidity', 'windspeed',  'hour', 'month', 'year','season_2', 'season_3', 'season_4', 'weather_2', 'weather_3', 'weather_4','dayofweek_1', 'dayofweek_2', 'dayofweek_3', 'dayofweek_4','dayofweek_5', 'dayofweek_6','temp_prev', 'atemp_prev', 'humidity_prev', 'windspeed_prev']]\n\npipe = pipelines[1][1]\n# print(pipe)\npipe.fit(X, np.log(train_Y[\"count\"]+1))\ny_pred = np.exp(pipe.predict(X_t))-1\n\nsub = test[[\"datetime\"]]\nsub[\"count\"] = y_pred\nsub.to_csv(\"Submission_1h.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:16:00.921342Z","iopub.execute_input":"2022-07-15T00:16:00.921676Z","iopub.status.idle":"2022-07-15T00:16:07.971961Z","shell.execute_reply.started":"2022-07-15T00:16:00.921644Z","shell.execute_reply":"2022-07-15T00:16:07.970939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 6.3 4 hour before (0.44363)\n\nX = train_X[['holiday', 'workingday', 'temp', 'atemp', 'humidity', 'windspeed', 'hour', 'month', 'year','season_2', 'season_3', 'season_4', 'weather_2', 'weather_3', 'weather_4','dayofweek_1', 'dayofweek_2', 'dayofweek_3', 'dayofweek_4','dayofweek_5', 'dayofweek_6','temp_prev_4h','atemp_prev_4h', 'humidity_prev_4h', 'windspeed_prev_4h']]\nX_t = test_X[['holiday', 'workingday', 'temp', 'atemp', 'humidity', 'windspeed', 'hour', 'month', 'year','season_2', 'season_3', 'season_4', 'weather_2', 'weather_3', 'weather_4','dayofweek_1', 'dayofweek_2', 'dayofweek_3', 'dayofweek_4','dayofweek_5', 'dayofweek_6','temp_prev_4h','atemp_prev_4h', 'humidity_prev_4h', 'windspeed_prev_4h']]\n\npipe = pipelines[1][1]\n# print(pipe)\npipe.fit(X, np.log(train_Y[\"count\"]+1))\ny_pred = np.exp(pipe.predict(X_t))-1\n\nsub = test[[\"datetime\"]]\nsub[\"count\"] = y_pred\nsub.to_csv(\"Submission_4h.csv\", index = False) ","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:16:07.973255Z","iopub.execute_input":"2022-07-15T00:16:07.973578Z","iopub.status.idle":"2022-07-15T00:16:15.812293Z","shell.execute_reply.started":"2022-07-15T00:16:07.973545Z","shell.execute_reply":"2022-07-15T00:16:15.811654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 6.4 12 hour before (Score: 0.45348)\n\nX = train_X[['holiday', 'workingday', 'temp', 'atemp', 'humidity', 'windspeed', 'hour', 'month', 'year','season_2', 'season_3', 'season_4', 'weather_2', 'weather_3', 'weather_4','dayofweek_1', 'dayofweek_2', 'dayofweek_3', 'dayofweek_4','dayofweek_5', 'dayofweek_6','temp_prev_12h','atemp_prev_12h', 'humidity_prev_12h', 'windspeed_prev_12h']]\nX_t = test_X[['holiday', 'workingday', 'temp', 'atemp', 'humidity', 'windspeed', 'hour', 'month', 'year','season_2', 'season_3', 'season_4', 'weather_2', 'weather_3', 'weather_4','dayofweek_1', 'dayofweek_2', 'dayofweek_3', 'dayofweek_4','dayofweek_5', 'dayofweek_6','temp_prev_12h','atemp_prev_12h', 'humidity_prev_12h', 'windspeed_prev_12h']]\n\npipe = pipelines[1][1]\n# print(pipe)\npipe.fit(X, np.log(train_Y[\"count\"]+1))\ny_pred = np.exp(pipe.predict(X_t))-1\n\nsub = test[[\"datetime\"]]\nsub[\"count\"] = y_pred\nsub.to_csv(\"Submission_12h.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:16:15.813260Z","iopub.execute_input":"2022-07-15T00:16:15.813643Z","iopub.status.idle":"2022-07-15T00:16:24.088335Z","shell.execute_reply.started":"2022-07-15T00:16:15.813583Z","shell.execute_reply":"2022-07-15T00:16:24.086288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 6.5 24 hour before (0.45218)\n\nX = train_X[['holiday', 'workingday', 'temp', 'atemp', 'humidity', 'windspeed', 'hour', 'month', 'year','season_2', 'season_3', 'season_4', 'weather_2', 'weather_3', 'weather_4','dayofweek_1', 'dayofweek_2', 'dayofweek_3', 'dayofweek_4','dayofweek_5', 'dayofweek_6','temp_prev_24h','atemp_prev_24h', 'humidity_prev_24h', 'windspeed_prev_24h']]\nX_t = test_X[['holiday', 'workingday', 'temp', 'atemp', 'humidity', 'windspeed', 'hour', 'month', 'year','season_2', 'season_3', 'season_4', 'weather_2', 'weather_3', 'weather_4','dayofweek_1', 'dayofweek_2', 'dayofweek_3', 'dayofweek_4','dayofweek_5', 'dayofweek_6','temp_prev_24h','atemp_prev_24h', 'humidity_prev_24h', 'windspeed_prev_24h']]\n\npipe = pipelines[1][1]\n# print(pipe)\npipe.fit(X, np.log(train_Y[\"count\"]+1))\ny_pred = np.exp(pipe.predict(X_t))-1\n\nsub = test[[\"datetime\"]]\nsub[\"count\"] = y_pred\nsub.to_csv(\"Submission_24h.csv\", index = False) ","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:16:24.089867Z","iopub.execute_input":"2022-07-15T00:16:24.090176Z","iopub.status.idle":"2022-07-15T00:16:37.479940Z","shell.execute_reply.started":"2022-07-15T00:16:24.090149Z","shell.execute_reply":"2022-07-15T00:16:37.479025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 6.6 mean of top 2 result (0.43576)\n\n\nsub_1h = pd.read_csv(\"Submission_1h.csv\") \nsub_4h = pd.read_csv(\"Submission_4h.csv\")\n\nsub = test[[\"datetime\"]]\nsub[\"count\"] = (sub_1h['count'] + sub_4h['count'])/2\nsub.to_csv(\"Submission_mix.csv\", index = False) ","metadata":{"execution":{"iopub.status.busy":"2022-07-15T00:16:37.481074Z","iopub.execute_input":"2022-07-15T00:16:37.481338Z","iopub.status.idle":"2022-07-15T00:16:37.537453Z","shell.execute_reply.started":"2022-07-15T00:16:37.481310Z","shell.execute_reply":"2022-07-15T00:16:37.536496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}