{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-14T09:00:54.797489Z","iopub.execute_input":"2022-07-14T09:00:54.799431Z","iopub.status.idle":"2022-07-14T09:00:55.449231Z","shell.execute_reply.started":"2022-07-14T09:00:54.799214Z","shell.execute_reply":"2022-07-14T09:00:55.448216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/bike-sharing-demand/train.csv')\ntest = pd.read_csv('/kaggle/input/bike-sharing-demand/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:00:55.451130Z","iopub.execute_input":"2022-07-14T09:00:55.451527Z","iopub.status.idle":"2022-07-14T09:00:55.510740Z","shell.execute_reply.started":"2022-07-14T09:00:55.451492Z","shell.execute_reply":"2022-07-14T09:00:55.509433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:00:55.512114Z","iopub.execute_input":"2022-07-14T09:00:55.512558Z","iopub.status.idle":"2022-07-14T09:00:55.535645Z","shell.execute_reply.started":"2022-07-14T09:00:55.512521Z","shell.execute_reply":"2022-07-14T09:00:55.534715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:00:55.537885Z","iopub.execute_input":"2022-07-14T09:00:55.538530Z","iopub.status.idle":"2022-07-14T09:00:55.553757Z","shell.execute_reply.started":"2022-07-14T09:00:55.538480Z","shell.execute_reply":"2022-07-14T09:00:55.552779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:00:55.555338Z","iopub.execute_input":"2022-07-14T09:00:55.555944Z","iopub.status.idle":"2022-07-14T09:00:55.578006Z","shell.execute_reply.started":"2022-07-14T09:00:55.555891Z","shell.execute_reply":"2022-07-14T09:00:55.577072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:00:55.579399Z","iopub.execute_input":"2022-07-14T09:00:55.579970Z","iopub.status.idle":"2022-07-14T09:00:55.593608Z","shell.execute_reply.started":"2022-07-14T09:00:55.579936Z","shell.execute_reply":"2022-07-14T09:00:55.592402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"outliers=np.abs(train['count']-train['count'].mean()) > (3.2*train['count'].std())\noutliers_num = len(train[outliers])\ntrain.drop(index=train[outliers].index)\nprint(\"delete \",outliers_num,\" outlier\")","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:00:55.594888Z","iopub.execute_input":"2022-07-14T09:00:55.595204Z","iopub.status.idle":"2022-07-14T09:00:55.606820Z","shell.execute_reply.started":"2022-07-14T09:00:55.595176Z","shell.execute_reply":"2022-07-14T09:00:55.605690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from datetime import datetime\nimport calendar\n\n#对时间进行提取\ndef time_process(df):\n    \n    #年、月、日、小时特征提取\n    df['year'] = pd.DatetimeIndex(df['datetime']).year\n    df['month'] = pd.DatetimeIndex(df['datetime']).month\n    df['day'] = pd.DatetimeIndex(df['datetime']).day\n    df['hour'] = pd.DatetimeIndex(df['datetime']).hour\n    \n    #探究工作日、双休日的特征\n    df['week'] = pd.DatetimeIndex(df['datetime']).weekofyear\n    df['weekday'] = pd.DatetimeIndex(df['datetime']).dayofweek\n    return df\n\ntrain = time_process(train)\ntest = time_process(test)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:00:55.608005Z","iopub.execute_input":"2022-07-14T09:00:55.609132Z","iopub.status.idle":"2022-07-14T09:00:55.658295Z","shell.execute_reply.started":"2022-07-14T09:00:55.609082Z","shell.execute_reply":"2022-07-14T09:00:55.657367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:00:55.659791Z","iopub.execute_input":"2022-07-14T09:00:55.661074Z","iopub.status.idle":"2022-07-14T09:00:55.687687Z","shell.execute_reply.started":"2022-07-14T09:00:55.661021Z","shell.execute_reply":"2022-07-14T09:00:55.686792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#尝试使用随机森林基于气温、季节插值\nfrom sklearn.ensemble import RandomForestClassifier\n\ndef wind_0_fill(df):\n    wind_0 = df[df['windspeed']==0]\n    wind_not0 = df[df['windspeed']!=0]\n    y_label = wind_not0['windspeed']\n    \n    #猜测风速和天气以及时间都有关\n    clf = RandomForestClassifier(n_estimators=500,max_depth=15,random_state=0)\n    windcolunms = ['season', 'weather', 'temp', 'atemp', 'humidity', 'hour', 'month']\n    clf.fit(wind_not0[windcolunms], y_label.astype('int'))\n    pred_y = clf.predict(wind_0[windcolunms])\n    \n    #预测结果填充\n    wind_0['windspeed'] = pred_y\n    df_rfw = wind_not0.append(wind_0)\n    df_rfw.reset_index(inplace=True)\n    return df_rfw\n\ntrain = wind_0_fill(train)\ntest = wind_0_fill(test)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:00:55.690926Z","iopub.execute_input":"2022-07-14T09:00:55.691511Z","iopub.status.idle":"2022-07-14T09:01:06.172517Z","shell.execute_reply.started":"2022-07-14T09:00:55.691474Z","shell.execute_reply":"2022-07-14T09:01:06.171176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 将行索引改为datetime\ndt = pd.DatetimeIndex(train['datetime'])\ntrain.set_index(dt, inplace=True)\ndtt = pd.DatetimeIndex(test['datetime'])\ntest.set_index(dtt, inplace=True)\ndef get_day(day_start):\n    day_end = day_start + pd.offsets.DateOffset(hours=23)\n    return pd.date_range(day_start, day_end, freq=\"H\")\n\n# 纳税日，仍需工作\ntrain.loc[get_day(pd.datetime(2011, 4, 15)), \"workingday\"] = 1\ntrain.loc[get_day(pd.datetime(2012, 4, 16)), \"workingday\"] = 1\n# 感恩节，不需要工作\ntest.loc[get_day(pd.datetime(2011, 11, 25)), \"workingday\"] = 0\ntest.loc[get_day(pd.datetime(2012, 11, 23)), \"workingday\"] = 0\n#圣诞节，不工作\ntest.loc[get_day(pd.datetime(2011, 12, 24)), \"workingday\"] = 0\ntest.loc[get_day(pd.datetime(2011, 12, 31)), \"workingday\"] = 0\ntest.loc[get_day(pd.datetime(2012, 12, 26)), \"workingday\"] = 0\ntest.loc[get_day(pd.datetime(2012, 12, 31)), \"workingday\"] = 0\n\n# 纳税日，不放假\ntrain.loc[get_day(pd.datetime(2011, 4, 15)), \"holiday\"] = 0\ntrain.loc[get_day(pd.datetime(2012, 4, 16)), \"holiday\"] = 0\n\n# 感恩节，放假\ntest.loc[get_day(pd.datetime(2011, 11, 25)), \"holiday\"] = 1\ntest.loc[get_day(pd.datetime(2012, 11, 23)), \"holiday\"] = 1\n#圣诞节，放假\ntest.loc[get_day(pd.datetime(2011, 12, 24)), \"holiday\"] = 1\ntest.loc[get_day(pd.datetime(2011, 12, 31)), \"holiday\"] = 1\ntest.loc[get_day(pd.datetime(2012, 12, 31)), \"holiday\"] = 1\n\n#暴雨\ntest.loc[get_day(pd.datetime(2012, 5, 21)), \"holiday\"] = 1\n#海啸\ntrain.loc[get_day(pd.datetime(2012, 6, 1)), \"holiday\"] = 1","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:01:06.174223Z","iopub.execute_input":"2022-07-14T09:01:06.175169Z","iopub.status.idle":"2022-07-14T09:01:06.216624Z","shell.execute_reply.started":"2022-07-14T09:01:06.175129Z","shell.execute_reply":"2022-07-14T09:01:06.215344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def name_process(df):\n    #季节、天气重命名，后续模型将用独热编码，易于可视化\n    df['season2'] = df['season']\n    df['weather2'] = df['weather']\n    df['season2'] = df['season2'].map({1:'Spring',2:'Summer',3:'Fall',4:'Winter'})\n    df['weather2'] = df['weather2'].map({1:'Clear',2:'Mist',3:'Light_Snow',4:'Heavy_Rain'})\n#     df['month'] = df['month'].map({1:'Jan',2:'Feb',3:'Mar',4:'Apr',5:'May',6:'Jun',7:'Jul',8:'Aug',9:'Sep',10:'Oct',11:'Nov',12:'Dec'})   \n    return df\n\ntrain = name_process(train)\ntest = name_process(test)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:01:06.218254Z","iopub.execute_input":"2022-07-14T09:01:06.218679Z","iopub.status.idle":"2022-07-14T09:01:06.236623Z","shell.execute_reply.started":"2022-07-14T09:01:06.218644Z","shell.execute_reply":"2022-07-14T09:01:06.235336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#不同时间段--使用量\nsns.boxplot(x='hour',y='count',data=train)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:01:06.238795Z","iopub.execute_input":"2022-07-14T09:01:06.239629Z","iopub.status.idle":"2022-07-14T09:01:06.754079Z","shell.execute_reply.started":"2022-07-14T09:01:06.239589Z","shell.execute_reply":"2022-07-14T09:01:06.752821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['peak'] = train[['hour', 'workingday']].apply(lambda x: (0, 1)[(x['workingday'] == 1 and  ( x['hour'] == 8 or 17 <= x['hour'] <= 18 or 12 <= x['hour'] <= 12)) or (x['workingday'] == 0 and  10 <= x['hour'] <= 19)], axis = 1)\ntest['peak'] = test[['hour', 'workingday']].apply(lambda x: (0, 1)[(x['workingday'] == 1 and  ( x['hour'] == 8 or 17 <= x['hour'] <= 18 or 12 <= x['hour'] <= 12)) or (x['workingday'] == 0 and  10 <= x['hour'] <= 19)], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:01:06.755816Z","iopub.execute_input":"2022-07-14T09:01:06.756575Z","iopub.status.idle":"2022-07-14T09:01:07.180198Z","shell.execute_reply.started":"2022-07-14T09:01:06.756524Z","shell.execute_reply":"2022-07-14T09:01:07.178949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#不同月份--使用量\nsns.boxplot(x='month',y='count',data=train)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:01:07.181633Z","iopub.execute_input":"2022-07-14T09:01:07.181972Z","iopub.status.idle":"2022-07-14T09:01:07.490805Z","shell.execute_reply.started":"2022-07-14T09:01:07.181942Z","shell.execute_reply":"2022-07-14T09:01:07.489612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#不同季节--使用量\nsns.boxplot(x='season',y='count',data=train)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:01:07.492219Z","iopub.execute_input":"2022-07-14T09:01:07.493171Z","iopub.status.idle":"2022-07-14T09:01:07.702135Z","shell.execute_reply.started":"2022-07-14T09:01:07.493133Z","shell.execute_reply":"2022-07-14T09:01:07.700804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#不同天气--使用量\nsns.catplot(x='weather',y='count',data=train)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:01:07.704081Z","iopub.execute_input":"2022-07-14T09:01:07.704747Z","iopub.status.idle":"2022-07-14T09:01:08.183882Z","shell.execute_reply.started":"2022-07-14T09:01:07.704699Z","shell.execute_reply":"2022-07-14T09:01:08.182518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#不同工作日--使用量\nsns.boxplot(x='weekday',y='count',data=train)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:01:08.185612Z","iopub.execute_input":"2022-07-14T09:01:08.186014Z","iopub.status.idle":"2022-07-14T09:01:08.432700Z","shell.execute_reply.started":"2022-07-14T09:01:08.185980Z","shell.execute_reply":"2022-07-14T09:01:08.431416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#是否工作日--使用量\nsns.boxplot(x='workingday',y='count',data=train)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:01:08.434470Z","iopub.execute_input":"2022-07-14T09:01:08.434863Z","iopub.status.idle":"2022-07-14T09:01:08.609774Z","shell.execute_reply.started":"2022-07-14T09:01:08.434830Z","shell.execute_reply":"2022-07-14T09:01:08.608863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#季节--使用量\nsns.pointplot(x='hour',y='count',hue='season',join=True,data=train)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:01:08.611219Z","iopub.execute_input":"2022-07-14T09:01:08.612366Z","iopub.status.idle":"2022-07-14T09:01:11.382657Z","shell.execute_reply.started":"2022-07-14T09:01:08.612292Z","shell.execute_reply":"2022-07-14T09:01:11.381400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#工作日--使用量\nsns.pointplot(x='hour',y='count',hue='weekday',join=True,data=train)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:01:11.383898Z","iopub.execute_input":"2022-07-14T09:01:11.384224Z","iopub.status.idle":"2022-07-14T09:01:16.071719Z","shell.execute_reply.started":"2022-07-14T09:01:11.384193Z","shell.execute_reply":"2022-07-14T09:01:16.070404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#注册用户--使用量\nsns.pointplot(x='hour',y='registered',hue=None,join=True,data=train)\nsns.pointplot(x='hour',y='casual',hue=None,join=True,data=train)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:01:16.073574Z","iopub.execute_input":"2022-07-14T09:01:16.074809Z","iopub.status.idle":"2022-07-14T09:01:17.726692Z","shell.execute_reply.started":"2022-07-14T09:01:16.074771Z","shell.execute_reply":"2022-07-14T09:01:17.725368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#温度--使用量\nfig = plt.subplots(figsize=(12,4))\nsns.regplot(x='temp',y='count',data=train)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:01:17.728244Z","iopub.execute_input":"2022-07-14T09:01:17.728629Z","iopub.status.idle":"2022-07-14T09:01:18.616395Z","shell.execute_reply.started":"2022-07-14T09:01:17.728596Z","shell.execute_reply":"2022-07-14T09:01:18.614770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#体感温度--使用量\nfig = plt.subplots(figsize=(12,4))\nsns.regplot(x='atemp',y='count',data=train)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:01:18.618078Z","iopub.execute_input":"2022-07-14T09:01:18.618752Z","iopub.status.idle":"2022-07-14T09:01:19.485883Z","shell.execute_reply.started":"2022-07-14T09:01:18.618707Z","shell.execute_reply":"2022-07-14T09:01:19.484510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#湿度--使用量\nfig = plt.subplots(figsize=(12,4))\nsns.regplot(x='humidity',y='count',data=train)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:01:19.487702Z","iopub.execute_input":"2022-07-14T09:01:19.488786Z","iopub.status.idle":"2022-07-14T09:01:20.362081Z","shell.execute_reply.started":"2022-07-14T09:01:19.488736Z","shell.execute_reply":"2022-07-14T09:01:20.360721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#风速--使用量\nfig = plt.subplots(figsize=(12,4))\nsns.regplot(x='windspeed',y='count',data=train)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:01:20.363900Z","iopub.execute_input":"2022-07-14T09:01:20.365233Z","iopub.status.idle":"2022-07-14T09:01:21.216965Z","shell.execute_reply.started":"2022-07-14T09:01:20.365181Z","shell.execute_reply":"2022-07-14T09:01:21.215532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#连续变量的相关性分析\ncorr = train[['temp','atemp','casual','registered','humidity','windspeed','count']].corr()\nmask = np.array(corr)\nmask[np.tril_indices_from(mask)] = False\nfig,ax = plt.subplots()\nfig.set_size_inches(13,7)\nsns.heatmap(corr,mask=mask,vmax=1,square=True,annot=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:01:21.218767Z","iopub.execute_input":"2022-07-14T09:01:21.219236Z","iopub.status.idle":"2022-07-14T09:01:21.558644Z","shell.execute_reply.started":"2022-07-14T09:01:21.219190Z","shell.execute_reply":"2022-07-14T09:01:21.557721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#样本正态分布情况一般\ntrain['count'].plot(kind='kde')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:01:21.562910Z","iopub.execute_input":"2022-07-14T09:01:21.563541Z","iopub.status.idle":"2022-07-14T09:01:22.058596Z","shell.execute_reply.started":"2022-07-14T09:01:21.563501Z","shell.execute_reply":"2022-07-14T09:01:22.057629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#对季节、天气、weekday等进行独热编码,并保留原属性编码，后续进行特征挑选\ntrain=pd.get_dummies(train,columns=['season2'])\ntrain=pd.get_dummies(train,columns=['weather2'])\n\ntest=pd.get_dummies(test,columns=['season2'])\ntest=pd.get_dummies(test,columns=['weather2'])","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:01:22.059911Z","iopub.execute_input":"2022-07-14T09:01:22.060555Z","iopub.status.idle":"2022-07-14T09:01:22.095894Z","shell.execute_reply.started":"2022-07-14T09:01:22.060506Z","shell.execute_reply":"2022-07-14T09:01:22.094604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"All_feature_columns = ['season','weather','temp','atemp','humidity','windspeed',\n                        'year','holiday','workingday','month','day','hour','week','weekday','peak',\n                       'season2_Fall','season2_Spring','season2_Summer','season2_Winter',\n                       'weather2_Clear','weather2_Heavy_Rain','weather2_Light_Snow','weather2_Mist']\n\nRFR_feature_columns = ['weather','temp','atemp','windspeed',\n                       'workingday','season','holiday',\n                       'hour','weekday','week','peak',\n                       'season2_Fall','season2_Spring','season2_Summer','season2_Winter',\n                      'weather2_Clear','weather2_Heavy_Rain','weather2_Light_Snow','weather2_Mist']\n\nGBR_feature_columns =['weather','temp','atemp','humidity','windspeed',\n                       'holiday','workingday','season',\n                       'hour','weekday','year',\n                      'season2_Fall','season2_Spring','season2_Summer','season2_Winter',\n                       'weather2_Clear','weather2_Heavy_Rain','weather2_Light_Snow','weather2_Mist']","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:01:22.097405Z","iopub.execute_input":"2022-07-14T09:01:22.097785Z","iopub.status.idle":"2022-07-14T09:01:22.104583Z","shell.execute_reply.started":"2022-07-14T09:01:22.097751Z","shell.execute_reply":"2022-07-14T09:01:22.103519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#拆分出训练数据\nRFR_X_train=train[RFR_feature_columns].values\nRFR_X_test=test[RFR_feature_columns].values\n\nGBR_X_train=train[GBR_feature_columns].values\nGBR_X_test=test[GBR_feature_columns].values\n\ny_casual=train['casual'].apply(lambda x: np.log1p(x)).values\ny_registered=train['registered'].apply(lambda x: np.log1p(x)).values\ny_count=train['count'].apply(lambda x: np.log1p(x)).values\n\nX_date=test['datetime'].values","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:01:22.106203Z","iopub.execute_input":"2022-07-14T09:01:22.107365Z","iopub.status.idle":"2022-07-14T09:01:22.179682Z","shell.execute_reply.started":"2022-07-14T09:01:22.107300Z","shell.execute_reply":"2022-07-14T09:01:22.178273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#评价标准RMSLE\ndef rmsle(y_real, y_pre):    \n    log1 = np.log(y_real+2)\n    log2 = np.log(y_pre+2)    \n    calc = (log1 - log2) ** 2\n    return np.sqrt(np.mean(calc))","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:01:22.181758Z","iopub.execute_input":"2022-07-14T09:01:22.182260Z","iopub.status.idle":"2022-07-14T09:01:22.188356Z","shell.execute_reply.started":"2022-07-14T09:01:22.182211Z","shell.execute_reply":"2022-07-14T09:01:22.187263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#拆分模型训练集\nfrom sklearn.model_selection import train_test_split\nX_train = train[All_feature_columns].values\nxd_train,xd_test,yd_train,yd_test = train_test_split(X_train,y_count,random_state=0)\n#对各种回归模型进行训练、调参、测试\n##LGBM\nfrom lightgbm import LGBMRegressor\ndef LGBM_model():\n    LGBM = LGBMRegressor(boosting_type='gbdt', objective='regression', num_leaves=1000,\n                                learning_rate=0.16, n_estimators=500, max_depth=15,\n                                metric='rmse', bagging_fraction=0.75, feature_fraction=0.75, reg_lambda=0.9)\n    LGBM.fit(xd_train, yd_train)\n    # 给出训练数据的预测值\n    pre_test = LGBM.predict(xd_test)\n    # 计算RMSLE\n    score = rmsle(yd_test,pre_test)\n    return score\n\n##随机森林\nfrom sklearn.ensemble import RandomForestRegressor\ndef RandomForest_model():\n    RFR = RandomForestRegressor(n_estimators = 1200, max_depth=15, random_state=0,n_jobs = -1)\n    RFR.fit(xd_train,yd_train)\n    # 给出训练数据的预测值\n    pre_test = RFR.predict(xd_test)\n    # 计算RMSLE\n    score = rmsle(yd_test,pre_test)\n    return score\n\n##决策树\nfrom sklearn.tree import DecisionTreeRegressor\ndef DecisionTree_model():\n    DTR = DecisionTreeRegressor(max_features='sqrt', splitter='random', min_samples_split=4, max_depth=15)\n    DTR.fit(xd_train,yd_train)\n    # 给出训练数据的预测值\n    pre_test = DTR.predict(xd_test)\n    # 计算RMSLE\n    score = rmsle(yd_test,pre_test)\n    return score\n\n##集成学习梯度提升决策树\nfrom sklearn.ensemble import GradientBoostingRegressor\ndef GradientBoosting_model():\n    GBR = GradientBoostingRegressor(n_estimators = 1200, max_depth = 15, random_state = 0)\n    GBR.fit(xd_train,yd_train)\n    # 给出训练数据的预测值\n    pre_test = GBR.predict(xd_test)\n    # 计算RMSLE\n    score = rmsle(yd_test,pre_test)\n    return score\n\n##逻辑斯蒂回归\nfrom sklearn.linear_model import LogisticRegression\ndef Logisic_model():\n    LG = LogisticRegression(penalty=\"l2\",tol=0.0001, C=1.0, solver= \"lbfgs\", max_iter=3000,multi_class='ovr', verbose=O)\n    LG.fit(xd_train,yd_train)\n    # 给出训练数据的预测值\n    pre_test = LG.predict(xd_test)\n    # 计算RMSLE\n    score = rmsle(yd_test,pre_test)\n    return score\n\n##AdaBoost\nfrom sklearn.ensemble import AdaBoostRegressor\ndef AdaBoost_model():\n    ABR = AdaBoostRegressor(learning_rate=0.1, loss='square', n_estimators=1000)\n    ABR.fit(xd_train,yd_train)\n    # 给出训练数据的预测值\n    pre_test = ABR.predict(xd_test)\n    # 计算RMSLE\n    score = rmsle(yd_test,pre_test)\n    return score","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:01:22.190155Z","iopub.execute_input":"2022-07-14T09:01:22.190919Z","iopub.status.idle":"2022-07-14T09:01:22.964781Z","shell.execute_reply.started":"2022-07-14T09:01:22.190873Z","shell.execute_reply":"2022-07-14T09:01:22.963432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#各种模型的RMSLE评价\nprint(\"LGBM_model:             \",LGBM_model())\nprint(\"RandomForest_model:     \",RandomForest_model())\nprint(\"DecisionTree_model:     \",DecisionTree_model())\nprint(\"GradientBoosting_model: \",GradientBoosting_model())\nprint(\"AdaBoost_model:         \",AdaBoost_model())","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:01:22.967575Z","iopub.execute_input":"2022-07-14T09:01:22.968119Z","iopub.status.idle":"2022-07-14T09:02:20.699077Z","shell.execute_reply.started":"2022-07-14T09:01:22.968067Z","shell.execute_reply":"2022-07-14T09:02:20.697767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#随机森林模型\nfrom sklearn.ensemble import RandomForestRegressor\nparams = {'n_estimators': 1000, \n          'max_depth': 15, \n          'random_state': 0, \n          'min_samples_split' : 5, \n          'n_jobs': -1}\n\nRFR1 = RandomForestRegressor(**params)\nRFR1.fit(RFR_X_train,y_casual)\nprint(\"model1拟合程度:\",RFR1.score(RFR_X_train,y_casual))\n\nRFR2 = RandomForestRegressor(**params)\nRFR2.fit(RFR_X_train,y_registered)\nprint(\"model2拟合程度:\",RFR2.score(RFR_X_train,y_registered))\n\nRFR3 = RandomForestRegressor(**params)\nRFR3.fit(RFR_X_train,y_count)\nprint(\"model3拟合程度:\",RFR3.score(RFR_X_train,y_count))","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:02:20.700609Z","iopub.execute_input":"2022-07-14T09:02:20.700964Z","iopub.status.idle":"2022-07-14T09:02:58.323369Z","shell.execute_reply.started":"2022-07-14T09:02:20.700926Z","shell.execute_reply":"2022-07-14T09:02:58.322135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#集成学习梯度提升决策树\nfrom sklearn.ensemble import GradientBoostingRegressor\n\nparams2 = {'n_estimators': 200, \n           'max_depth': 5, \n           'random_state': 0, \n           'min_samples_leaf' : 10, \n           'learning_rate': 0.15, \n           'subsample': 0.7, \n           'loss': 'ls'}\n\nGBR1 = GradientBoostingRegressor(**params2)\nGBR1.fit(GBR_X_train,y_casual)\nprint(\"model拟合程度:\",GBR1.score(GBR_X_train,y_casual))\n\nGBR2 = GradientBoostingRegressor(**params2)\nGBR2.fit(GBR_X_train,y_registered)\nprint(\"model拟合程度:\",GBR2.score(GBR_X_train,y_registered))\n\nGBR3 = GradientBoostingRegressor(**params2)\nGBR3.fit(GBR_X_train,y_count)\nprint(\"model拟合程度:\",GBR3.score(GBR_X_train,y_count))","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:02:58.327860Z","iopub.execute_input":"2022-07-14T09:02:58.328370Z","iopub.status.idle":"2022-07-14T09:03:06.458208Z","shell.execute_reply.started":"2022-07-14T09:02:58.328302Z","shell.execute_reply":"2022-07-14T09:03:06.456685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RFR_pre_casual = RFR1.predict(RFR_X_test)\nRFR_pre_casual=np.exp(RFR_pre_casual)-1\nRFR_pre_registered = RFR2.predict(RFR_X_test)\nRFR_pre_registered=np.exp(RFR_pre_registered)-1\nRFR_pre = RFR_pre_casual+RFR_pre_registered\n\nGBR_pre_casual = GBR1.predict(GBR_X_test)\nGBR_pre_casual=np.exp(GBR_pre_casual)-1\nGBR_pre_registered = GBR2.predict(GBR_X_test)\nGBR_pre_registered=np.exp(GBR_pre_registered)-1\nGBR_pre = GBR_pre_casual+GBR_pre_registered\n\nsubmit1 = pd.DataFrame({'datetime':X_date,'count':0.3*RFR_pre+0.7*GBR_pre})\nsubmit1.to_csv('/kaggle/working/submisssion.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:06:00.111750Z","iopub.execute_input":"2022-07-14T09:06:00.112217Z","iopub.status.idle":"2022-07-14T09:06:01.396061Z","shell.execute_reply.started":"2022-07-14T09:06:00.112172Z","shell.execute_reply":"2022-07-14T09:06:01.394634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RFR_pre_count = RFR3.predict(RFR_X_test)\nRFR_pre_count = np.exp(RFR_pre_count)-1\nGBR_pre_count = GBR3.predict(GBR_X_test)\nGBR_pre_count = np.exp(GBR_pre_count)-1\n\npre_count=0.3*RFR_pre_count+0.7*GBR_pre_count\nsubmit2 = pd.DataFrame({'datetime':X_date,'count':pre_count})\nsubmit2.to_csv('/kaggle/working/submisssion_2.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:06:11.231436Z","iopub.execute_input":"2022-07-14T09:06:11.231877Z","iopub.status.idle":"2022-07-14T09:06:11.887335Z","shell.execute_reply.started":"2022-07-14T09:06:11.231840Z","shell.execute_reply":"2022-07-14T09:06:11.885748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}