{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#导入基本的库\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport warnings\nwarnings.filterwarnings(\"ignore\")#忽略警告","metadata":{"ExecuteTime":{"end_time":"2022-05-26T04:50:51.839508Z","start_time":"2022-05-26T04:50:47.392782Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:15.331363Z","iopub.execute_input":"2022-06-04T09:05:15.332246Z","iopub.status.idle":"2022-06-04T09:05:15.337165Z","shell.execute_reply.started":"2022-06-04T09:05:15.332208Z","shell.execute_reply":"2022-06-04T09:05:15.336326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#读取数据\ntrain = pd.read_csv('/kaggle/input/bike-sharing-demand/train.csv')\ntest = pd.read_csv('/kaggle/input/bike-sharing-demand/test.csv')","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:03:57.648729Z","start_time":"2022-05-25T13:03:57.447134Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:15.354202Z","iopub.execute_input":"2022-06-04T09:05:15.35495Z","iopub.status.idle":"2022-06-04T09:05:15.390991Z","shell.execute_reply.started":"2022-06-04T09:05:15.354916Z","shell.execute_reply":"2022-06-04T09:05:15.390176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#数据总览\ntrain.head()\n# test.head()\ntrain.info()\n# test.info()\n# train.describe()\n# test.describe()","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:03:57.704126Z","start_time":"2022-05-25T13:03:57.651837Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:15.393048Z","iopub.execute_input":"2022-06-04T09:05:15.393457Z","iopub.status.idle":"2022-06-04T09:05:15.408688Z","shell.execute_reply.started":"2022-06-04T09:05:15.393417Z","shell.execute_reply":"2022-06-04T09:05:15.407681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#离群点共147个，去掉\noutliers=np.abs(train['count']-train['count'].mean()) > (3*train['count'].std())\noutliers_num = len(train[outliers])\ntrain.drop(index=train[outliers].index)\nprint(\"已删除\",outliers_num,\"个离群点\")","metadata":{"execution":{"iopub.status.busy":"2022-06-04T09:05:15.429684Z","iopub.execute_input":"2022-06-04T09:05:15.430132Z","iopub.status.idle":"2022-06-04T09:05:15.441032Z","shell.execute_reply.started":"2022-06-04T09:05:15.4301Z","shell.execute_reply":"2022-06-04T09:05:15.439912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from datetime import datetime\nimport calendar\n#对时间进行提取\ndef time_process(df):\n    #年、月、日、小时特征提取\n    df['year'] = pd.DatetimeIndex(df['datetime']).year\n    df['month'] = pd.DatetimeIndex(df['datetime']).month\n    df['day'] = pd.DatetimeIndex(df['datetime']).day\n    df['hour'] = pd.DatetimeIndex(df['datetime']).hour\n    #将日期的礼拜数标出，以探究工作日、双休日的特征\n    df['week'] = pd.DatetimeIndex(df['datetime']).weekofyear\n    df['weekday'] = pd.DatetimeIndex(df['datetime']).dayofweek\n    return df\n\ntrain = time_process(train)\ntest = time_process(test)","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:03:58.083139Z","start_time":"2022-05-25T13:03:57.707311Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:15.478577Z","iopub.execute_input":"2022-06-04T09:05:15.47899Z","iopub.status.idle":"2022-06-04T09:05:15.544635Z","shell.execute_reply.started":"2022-06-04T09:05:15.478957Z","shell.execute_reply":"2022-06-04T09:05:15.543908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2022-06-04T09:05:15.546269Z","iopub.execute_input":"2022-06-04T09:05:15.547149Z","iopub.status.idle":"2022-06-04T09:05:15.571896Z","shell.execute_reply.started":"2022-06-04T09:05:15.547107Z","shell.execute_reply":"2022-06-04T09:05:15.570856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#风速有很多缺失值，尝试使用随机森林基于气温、季节插值\nfrom sklearn.ensemble import RandomForestClassifier\ndef wind_0_fill(df):\n    wind_0 = df[df['windspeed']==0]\n    wind_not0 = df[df['windspeed']!=0]\n    y_label = wind_not0['windspeed']\n    #猜测风速和天气以及时间都有关\n    clf = RandomForestClassifier(n_estimators=1000,max_depth=10,random_state=0)\n    windcolunms = ['season', 'weather', 'temp', 'atemp', 'humidity', 'hour', 'month']\n    clf.fit(wind_not0[windcolunms], y_label.astype('int'))\n    pred_y = clf.predict(wind_0[windcolunms])\n    #预测结果填充\n    wind_0['windspeed'] = pred_y\n    df_rfw = wind_not0.append(wind_0)\n    df_rfw.reset_index(inplace=True)\n    return df_rfw\n\ntrain = wind_0_fill(train)\ntest = wind_0_fill(test)","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:04:14.716136Z","start_time":"2022-05-25T13:03:58.086361Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:15.573449Z","iopub.execute_input":"2022-06-04T09:05:15.574432Z","iopub.status.idle":"2022-06-04T09:05:28.036314Z","shell.execute_reply.started":"2022-06-04T09:05:15.574392Z","shell.execute_reply":"2022-06-04T09:05:28.035251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 将行索引改为datetime\ndt = pd.DatetimeIndex(train['datetime'])\ntrain.set_index(dt, inplace=True)\ndtt = pd.DatetimeIndex(test['datetime'])\ntest.set_index(dtt, inplace=True)\ndef get_day(day_start):\n    day_end = day_start + pd.offsets.DateOffset(hours=23)\n    return pd.date_range(day_start, day_end, freq=\"H\")\n\n# 纳税日，仍需工作\ntrain.loc[get_day(pd.datetime(2011, 4, 15)), \"workingday\"] = 1\ntrain.loc[get_day(pd.datetime(2012, 4, 16)), \"workingday\"] = 1\n# 感恩节，不需要工作\ntest.loc[get_day(pd.datetime(2011, 11, 25)), \"workingday\"] = 0\ntest.loc[get_day(pd.datetime(2012, 11, 23)), \"workingday\"] = 0\n#圣诞节，不工作\ntest.loc[get_day(pd.datetime(2011, 12, 24)), \"workingday\"] = 0\ntest.loc[get_day(pd.datetime(2011, 12, 31)), \"workingday\"] = 0\ntest.loc[get_day(pd.datetime(2012, 12, 26)), \"workingday\"] = 0\ntest.loc[get_day(pd.datetime(2012, 12, 31)), \"workingday\"] = 0\n\n# 纳税日，不放假\ntrain.loc[get_day(pd.datetime(2011, 4, 15)), \"holiday\"] = 0\ntrain.loc[get_day(pd.datetime(2012, 4, 16)), \"holiday\"] = 0\n\n# 感恩节，放假\ntest.loc[get_day(pd.datetime(2011, 11, 25)), \"holiday\"] = 1\ntest.loc[get_day(pd.datetime(2012, 11, 23)), \"holiday\"] = 1\n#圣诞节，放假\ntest.loc[get_day(pd.datetime(2011, 12, 24)), \"holiday\"] = 1\ntest.loc[get_day(pd.datetime(2011, 12, 31)), \"holiday\"] = 1\ntest.loc[get_day(pd.datetime(2012, 12, 31)), \"holiday\"] = 1\n\n#暴雨\ntest.loc[get_day(pd.datetime(2012, 5, 21)), \"holiday\"] = 1\n#海啸\ntrain.loc[get_day(pd.datetime(2012, 6, 1)), \"holiday\"] = 1","metadata":{"execution":{"iopub.status.busy":"2022-06-04T09:05:28.038334Z","iopub.execute_input":"2022-06-04T09:05:28.038742Z","iopub.status.idle":"2022-06-04T09:05:28.086766Z","shell.execute_reply.started":"2022-06-04T09:05:28.038694Z","shell.execute_reply":"2022-06-04T09:05:28.085968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def name_process(df):\n    #季节、天气重命名，后续模型将用独热编码，易于可视化\n    df['season2'] = df['season']\n    df['weather2'] = df['weather']\n    df['season2'] = df['season2'].map({1:'Spring',2:'Summer',3:'Fall',4:'Winter'})\n    df['weather2'] = df['weather2'].map({1:'Clear',2:'Mist',3:'Light_Snow',4:'Heavy_Rain'})\n#     df['month'] = df['month'].map({1:'Jan',2:'Feb',3:'Mar',4:'Apr',5:'May',6:'Jun',7:'Jul',8:'Aug',9:'Sep',10:'Oct',11:'Nov',12:'Dec'})   \n    return df\n\ntrain = name_process(train)\ntest = name_process(test)","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:04:14.745416Z","start_time":"2022-05-25T13:04:14.720125Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:28.087855Z","iopub.execute_input":"2022-06-04T09:05:28.08817Z","iopub.status.idle":"2022-06-04T09:05:28.101326Z","shell.execute_reply.started":"2022-06-04T09:05:28.088142Z","shell.execute_reply":"2022-06-04T09:05:28.100482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#可视化分析\n#不同时间段--使用量\nsns.boxplot(x='hour',y='count',data=train)","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:04:15.584906Z","start_time":"2022-05-25T13:04:14.804326Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:28.102652Z","iopub.execute_input":"2022-06-04T09:05:28.103225Z","iopub.status.idle":"2022-06-04T09:05:28.598174Z","shell.execute_reply.started":"2022-06-04T09:05:28.103194Z","shell.execute_reply":"2022-06-04T09:05:28.597204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#根据数据，解译出使用量高峰时段\ntrain['peak'] = train[['hour', 'workingday']].apply(lambda x: (0, 1)[(x['workingday'] == 1 and  ( x['hour'] == 8 or 17 <= x['hour'] <= 18 or 12 <= x['hour'] <= 12)) or (x['workingday'] == 0 and  10 <= x['hour'] <= 19)], axis = 1)\ntest['peak'] = test[['hour', 'workingday']].apply(lambda x: (0, 1)[(x['workingday'] == 1 and  ( x['hour'] == 8 or 17 <= x['hour'] <= 18 or 12 <= x['hour'] <= 12)) or (x['workingday'] == 0 and  10 <= x['hour'] <= 19)], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-06-04T09:05:28.599717Z","iopub.execute_input":"2022-06-04T09:05:28.600309Z","iopub.status.idle":"2022-06-04T09:05:29.21803Z","shell.execute_reply.started":"2022-06-04T09:05:28.600265Z","shell.execute_reply":"2022-06-04T09:05:29.217241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#不同月份--使用量\nsns.boxplot(x='month',y='count',data=train)","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:04:15.844843Z","start_time":"2022-05-25T13:04:15.586903Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:29.219276Z","iopub.execute_input":"2022-06-04T09:05:29.219686Z","iopub.status.idle":"2022-06-04T09:05:29.534819Z","shell.execute_reply.started":"2022-06-04T09:05:29.219647Z","shell.execute_reply":"2022-06-04T09:05:29.534229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#不同季节--使用量\nsns.boxplot(x='season',y='count',data=train)","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:04:15.996961Z","start_time":"2022-05-25T13:04:15.85038Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:29.535811Z","iopub.execute_input":"2022-06-04T09:05:29.536475Z","iopub.status.idle":"2022-06-04T09:05:29.740617Z","shell.execute_reply.started":"2022-06-04T09:05:29.536442Z","shell.execute_reply":"2022-06-04T09:05:29.739874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#不同天气--使用量\nsns.catplot(x='weather',y='count',data=train)","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:04:16.298855Z","start_time":"2022-05-25T13:04:15.997959Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:29.743229Z","iopub.execute_input":"2022-06-04T09:05:29.743683Z","iopub.status.idle":"2022-06-04T09:05:30.061021Z","shell.execute_reply.started":"2022-06-04T09:05:29.74365Z","shell.execute_reply":"2022-06-04T09:05:30.060332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#不同工作日--使用量\nsns.boxplot(x='weekday',y='count',data=train)","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:04:16.496905Z","start_time":"2022-05-25T13:04:16.301908Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:30.061915Z","iopub.execute_input":"2022-06-04T09:05:30.062655Z","iopub.status.idle":"2022-06-04T09:05:30.314573Z","shell.execute_reply.started":"2022-06-04T09:05:30.062623Z","shell.execute_reply":"2022-06-04T09:05:30.313816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#是否工作日--使用量\nsns.boxplot(x='workingday',y='count',data=train)","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:04:16.62711Z","start_time":"2022-05-25T13:04:16.498792Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:30.31565Z","iopub.execute_input":"2022-06-04T09:05:30.315964Z","iopub.status.idle":"2022-06-04T09:05:30.489473Z","shell.execute_reply.started":"2022-06-04T09:05:30.315939Z","shell.execute_reply":"2022-06-04T09:05:30.488592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#季节--使用量\nsns.pointplot(x='hour',y='count',hue='season',join=True,data=train)","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:04:19.707024Z","start_time":"2022-05-25T13:04:16.629017Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:30.490873Z","iopub.execute_input":"2022-06-04T09:05:30.4913Z","iopub.status.idle":"2022-06-04T09:05:34.014095Z","shell.execute_reply.started":"2022-06-04T09:05:30.491259Z","shell.execute_reply":"2022-06-04T09:05:34.01325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#工作日--使用量\nsns.pointplot(x='hour',y='count',hue='weekday',join=True,data=train)","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:04:25.102624Z","start_time":"2022-05-25T13:04:19.709018Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:34.015316Z","iopub.execute_input":"2022-06-04T09:05:34.015595Z","iopub.status.idle":"2022-06-04T09:05:39.951565Z","shell.execute_reply.started":"2022-06-04T09:05:34.015569Z","shell.execute_reply":"2022-06-04T09:05:39.9506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#注册用户--使用量\nsns.pointplot(x='hour',y='registered',hue=None,join=True,data=train)\nsns.pointplot(x='hour',y='casual',hue=None,join=True,data=train)","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:04:26.904869Z","start_time":"2022-05-25T13:04:25.10549Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:39.953211Z","iopub.execute_input":"2022-06-04T09:05:39.95365Z","iopub.status.idle":"2022-06-04T09:05:41.985867Z","shell.execute_reply.started":"2022-06-04T09:05:39.953607Z","shell.execute_reply":"2022-06-04T09:05:41.985022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#温度--使用量\nfig = plt.subplots(figsize=(12,4))\nsns.regplot(x='temp',y='count',data=train)","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:04:29.398915Z","start_time":"2022-05-25T13:04:26.908887Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:41.987051Z","iopub.execute_input":"2022-06-04T09:05:41.987449Z","iopub.status.idle":"2022-06-04T09:05:42.928982Z","shell.execute_reply.started":"2022-06-04T09:05:41.987418Z","shell.execute_reply":"2022-06-04T09:05:42.927991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#体感温度--使用量\nfig = plt.subplots(figsize=(12,4))\nsns.regplot(x='atemp',y='count',data=train)","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:04:32.010842Z","start_time":"2022-05-25T13:04:29.401075Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:42.93177Z","iopub.execute_input":"2022-06-04T09:05:42.93229Z","iopub.status.idle":"2022-06-04T09:05:44.128106Z","shell.execute_reply.started":"2022-06-04T09:05:42.932258Z","shell.execute_reply":"2022-06-04T09:05:44.127271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#湿度--使用量\nfig = plt.subplots(figsize=(12,4))\nsns.regplot(x='humidity',y='count',data=train)","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:04:34.868518Z","start_time":"2022-05-25T13:04:32.041975Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:44.129403Z","iopub.execute_input":"2022-06-04T09:05:44.12968Z","iopub.status.idle":"2022-06-04T09:05:45.022699Z","shell.execute_reply.started":"2022-06-04T09:05:44.129655Z","shell.execute_reply":"2022-06-04T09:05:45.021595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#风速--使用量\nfig = plt.subplots(figsize=(12,4))\nsns.regplot(x='windspeed',y='count',data=train)","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:04:37.776015Z","start_time":"2022-05-25T13:04:34.871944Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:45.02445Z","iopub.execute_input":"2022-06-04T09:05:45.0249Z","iopub.status.idle":"2022-06-04T09:05:45.914647Z","shell.execute_reply.started":"2022-06-04T09:05:45.024858Z","shell.execute_reply":"2022-06-04T09:05:45.913804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#连续变量的相关性分析\ncorr = train[['temp','atemp','casual','registered','humidity','windspeed','count']].corr()\nmask = np.array(corr)\nmask[np.tril_indices_from(mask)] = False\nfig,ax = plt.subplots()\nfig.set_size_inches(15,8)\nsns.heatmap(corr,mask=mask,vmax=.8,square=True,annot=True)","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:04:38.093934Z","start_time":"2022-05-25T13:04:37.800695Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:45.915906Z","iopub.execute_input":"2022-06-04T09:05:45.916321Z","iopub.status.idle":"2022-06-04T09:05:46.246479Z","shell.execute_reply.started":"2022-06-04T09:05:45.916291Z","shell.execute_reply":"2022-06-04T09:05:46.245584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#样本正态分布情况一般\ntrain['count'].plot(kind='kde')","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:04:38.574997Z","start_time":"2022-05-25T13:04:38.097694Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:46.247868Z","iopub.execute_input":"2022-06-04T09:05:46.248475Z","iopub.status.idle":"2022-06-04T09:05:46.683465Z","shell.execute_reply.started":"2022-06-04T09:05:46.248433Z","shell.execute_reply":"2022-06-04T09:05:46.68258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#进行log1p变换\nimport math\ntrain['count_log']=train['count'].apply(lambda x: math.log(x+1))\ntrain['count_log'].plot(kind='kde')","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:04:38.953415Z","start_time":"2022-05-25T13:04:38.578988Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:46.684736Z","iopub.execute_input":"2022-06-04T09:05:46.68515Z","iopub.status.idle":"2022-06-04T09:05:47.090256Z","shell.execute_reply.started":"2022-06-04T09:05:46.685094Z","shell.execute_reply":"2022-06-04T09:05:47.08966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#对季节、天气、weekday等进行独热编码,并保留原属性编码，后续进行特征挑选\ntrain=pd.get_dummies(train,columns=['season2'])\ntrain=pd.get_dummies(train,columns=['weather2'])\n\ntest=pd.get_dummies(test,columns=['season2'])\ntest=pd.get_dummies(test,columns=['weather2'])","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:04:39.109018Z","start_time":"2022-05-25T13:04:38.985048Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:47.091265Z","iopub.execute_input":"2022-06-04T09:05:47.09188Z","iopub.status.idle":"2022-06-04T09:05:47.122003Z","shell.execute_reply.started":"2022-06-04T09:05:47.091849Z","shell.execute_reply":"2022-06-04T09:05:47.121149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"All_feature_columns = ['season','weather','temp','atemp','humidity','windspeed',\n                        'year','holiday','workingday','month','day','hour','week','weekday','peak',\n                       'season2_Fall','season2_Spring','season2_Summer','season2_Winter',\n                       'weather2_Clear','weather2_Heavy_Rain','weather2_Light_Snow','weather2_Mist']\n\nRFR_feature_columns = ['weather','temp','atemp','windspeed',\n                       'workingday','season','holiday',\n                       'hour','weekday','week','peak',\n                       'season2_Fall','season2_Spring','season2_Summer','season2_Winter',\n                      'weather2_Clear','weather2_Heavy_Rain','weather2_Light_Snow','weather2_Mist']\n\nGBR_feature_columns =['weather','temp','atemp','humidity','windspeed',\n                       'holiday','workingday','season',\n                       'hour','weekday','year',\n                      'season2_Fall','season2_Spring','season2_Summer','season2_Winter',\n                       'weather2_Clear','weather2_Heavy_Rain','weather2_Light_Snow','weather2_Mist']","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:04:39.427637Z","start_time":"2022-05-25T13:04:39.409742Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:47.12329Z","iopub.execute_input":"2022-06-04T09:05:47.123693Z","iopub.status.idle":"2022-06-04T09:05:47.131209Z","shell.execute_reply.started":"2022-06-04T09:05:47.123653Z","shell.execute_reply":"2022-06-04T09:05:47.130358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#拆分出训练数据\nRFR_X_train=train[RFR_feature_columns].values\nRFR_X_test=test[RFR_feature_columns].values\n\nGBR_X_train=train[GBR_feature_columns].values\nGBR_X_test=test[GBR_feature_columns].values\n\ny_casual=train['casual'].apply(lambda x: np.log1p(x)).values\ny_registered=train['registered'].apply(lambda x: np.log1p(x)).values\ny_count=train['count'].apply(lambda x: np.log1p(x)).values\n\nX_date=test['datetime'].values","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:04:39.519326Z","start_time":"2022-05-25T13:04:39.429632Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:47.132277Z","iopub.execute_input":"2022-06-04T09:05:47.132571Z","iopub.status.idle":"2022-06-04T09:05:47.227794Z","shell.execute_reply.started":"2022-06-04T09:05:47.132547Z","shell.execute_reply":"2022-06-04T09:05:47.226963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#评价标准RMSLE\ndef rmsle(y_real, y_pre):    \n    log1 = np.log(y_real+1)\n    log2 = np.log(y_pre+1)    \n    calc = (log1 - log2) ** 2\n    return np.sqrt(np.mean(calc))","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:04:39.581183Z","start_time":"2022-05-25T13:04:39.568218Z"},"execution":{"iopub.status.busy":"2022-06-04T09:05:47.229287Z","iopub.execute_input":"2022-06-04T09:05:47.229806Z","iopub.status.idle":"2022-06-04T09:05:47.234757Z","shell.execute_reply.started":"2022-06-04T09:05:47.229776Z","shell.execute_reply":"2022-06-04T09:05:47.234096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#拆分模型训练集\nfrom sklearn.model_selection import train_test_split\nX_train = train[All_feature_columns].values\nxd_train,xd_test,yd_train,yd_test = train_test_split(X_train,y_count,random_state=0)\n#对各种回归模型进行训练、调参、测试\n##LGBM\nfrom lightgbm import LGBMRegressor\ndef LGBM_model():\n    LGBM = LGBMRegressor(boosting_type='gbdt', objective='regression', num_leaves=1200,\n                                learning_rate=0.17, n_estimators=1000, max_depth=10,\n                                metric='rmse', bagging_fraction=0.8, feature_fraction=0.8, reg_lambda=0.9)\n    LGBM.fit(xd_train, yd_train)\n    # 给出训练数据的预测值\n    pre_test = LGBM.predict(xd_test)\n    # 计算RMSLE\n    score = rmsle(yd_test,pre_test)\n    return score\n\n##随机森林\nfrom sklearn.ensemble import RandomForestRegressor\ndef RandomForest_model():\n    RFR = RandomForestRegressor(n_estimators = 1000, max_depth=15, random_state=0,n_jobs = -1)\n    RFR.fit(xd_train,yd_train)\n    # 给出训练数据的预测值\n    pre_test = RFR.predict(xd_test)\n    # 计算RMSLE\n    score = rmsle(yd_test,pre_test)\n    return score\n\n##决策树\nfrom sklearn.tree import DecisionTreeRegressor\ndef DecisionTree_model():\n    DTR = DecisionTreeRegressor(max_features='sqrt', splitter='random', min_samples_split=4, max_depth=10)\n    DTR.fit(xd_train,yd_train)\n    # 给出训练数据的预测值\n    pre_test = DTR.predict(xd_test)\n    # 计算RMSLE\n    score = rmsle(yd_test,pre_test)\n    return score\n\n##集成学习梯度提升决策树\nfrom sklearn.ensemble import GradientBoostingRegressor\ndef GradientBoosting_model():\n    GBR = GradientBoostingRegressor(n_estimators = 1000, max_depth = 5, random_state = 0)\n    GBR.fit(xd_train,yd_train)\n    # 给出训练数据的预测值\n    pre_test = GBR.predict(xd_test)\n    # 计算RMSLE\n    score = rmsle(yd_test,pre_test)\n    return score\n\n##逻辑斯蒂回归\nfrom sklearn.linear_model import LogisticRegression\ndef Logisic_model():\n    LG = LogisticRegression(penalty=\"l2\",tol=0.0001, C=1.0, solver= \"lbfgs\", max_iter=3000,multi_class='ovr', verbose=O)\n    LG.fit(xd_train,yd_train)\n    # 给出训练数据的预测值\n    pre_test = LG.predict(xd_test)\n    # 计算RMSLE\n    score = rmsle(yd_test,pre_test)\n    return score\n\n##AdaBoost\nfrom sklearn.ensemble import AdaBoostRegressor\ndef AdaBoost_model():\n    ABR = AdaBoostRegressor(learning_rate=0.1, loss='square', n_estimators=1000)\n    ABR.fit(xd_train,yd_train)\n    # 给出训练数据的预测值\n    pre_test = ABR.predict(xd_test)\n    # 计算RMSLE\n    score = rmsle(yd_test,pre_test)\n    return score","metadata":{"execution":{"iopub.status.busy":"2022-06-04T09:05:47.236131Z","iopub.execute_input":"2022-06-04T09:05:47.236534Z","iopub.status.idle":"2022-06-04T09:05:47.258926Z","shell.execute_reply.started":"2022-06-04T09:05:47.236505Z","shell.execute_reply":"2022-06-04T09:05:47.258049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#各种模型的RMSLE评价\nprint(\"LGBM_model:             \",LGBM_model())\nprint(\"RandomForest_model:     \",RandomForest_model())\nprint(\"DecisionTree_model:     \",DecisionTree_model())\nprint(\"GradientBoosting_model: \",GradientBoosting_model())\nprint(\"AdaBoost_model:         \",AdaBoost_model())","metadata":{"execution":{"iopub.status.busy":"2022-06-04T09:05:47.262952Z","iopub.execute_input":"2022-06-04T09:05:47.263591Z","iopub.status.idle":"2022-06-04T09:06:33.158242Z","shell.execute_reply.started":"2022-06-04T09:05:47.263558Z","shell.execute_reply":"2022-06-04T09:06:33.156948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#随机森林模型\nfrom sklearn.ensemble import RandomForestRegressor\nparams = {'n_estimators': 1000, \n          'max_depth': 15, \n          'random_state': 0, \n          'min_samples_split' : 5, \n          'n_jobs': -1}\n\nRFR1 = RandomForestRegressor(**params)\nRFR1.fit(RFR_X_train,y_casual)\nprint(\"model拟合程度:\",RFR1.score(RFR_X_train,y_casual))\n\nRFR2 = RandomForestRegressor(**params)\nRFR2.fit(RFR_X_train,y_registered)\nprint(\"model拟合程度:\",RFR2.score(RFR_X_train,y_registered))\n\nRFR3 = RandomForestRegressor(**params)\nRFR3.fit(RFR_X_train,y_count)\nprint(\"model拟合程度:\",RFR3.score(RFR_X_train,y_count))","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:11:38.03917Z","start_time":"2022-05-25T13:06:09.483007Z"},"execution":{"iopub.status.busy":"2022-06-04T09:06:33.160067Z","iopub.execute_input":"2022-06-04T09:06:33.160623Z","iopub.status.idle":"2022-06-04T09:07:09.978518Z","shell.execute_reply.started":"2022-06-04T09:06:33.16059Z","shell.execute_reply":"2022-06-04T09:07:09.977537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#集成学习梯度提升决策树\nfrom sklearn.ensemble import GradientBoostingRegressor\n\nparams2 = {'n_estimators': 150, \n           'max_depth': 5, \n           'random_state': 0, \n           'min_samples_leaf' : 10, \n           'learning_rate': 0.1, \n           'subsample': 0.7, \n           'loss': 'ls'}\n\nGBR1 = GradientBoostingRegressor(**params2)\nGBR1.fit(GBR_X_train,y_casual)\nprint(\"model拟合程度:\",GBR1.score(GBR_X_train,y_casual))\n\nGBR2 = GradientBoostingRegressor(**params2)\nGBR2.fit(GBR_X_train,y_registered)\nprint(\"model拟合程度:\",GBR2.score(GBR_X_train,y_registered))\n\nGBR3 = GradientBoostingRegressor(**params2)\nGBR3.fit(GBR_X_train,y_count)\nprint(\"model拟合程度:\",GBR3.score(GBR_X_train,y_count))","metadata":{"execution":{"iopub.status.busy":"2022-06-04T09:07:09.979978Z","iopub.execute_input":"2022-06-04T09:07:09.980401Z","iopub.status.idle":"2022-06-04T09:07:15.881493Z","shell.execute_reply.started":"2022-06-04T09:07:09.980359Z","shell.execute_reply":"2022-06-04T09:07:15.880121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RFR_pre_casual = RFR1.predict(RFR_X_test)\nRFR_pre_casual=np.exp(RFR_pre_casual)-1\nRFR_pre_registered = RFR2.predict(RFR_X_test)\nRFR_pre_registered=np.exp(RFR_pre_registered)-1\nRFR_pre = RFR_pre_casual+RFR_pre_registered\n\nGBR_pre_casual = GBR1.predict(GBR_X_test)\nGBR_pre_casual=np.exp(GBR_pre_casual)-1\nGBR_pre_registered = GBR2.predict(GBR_X_test)\nGBR_pre_registered=np.exp(GBR_pre_registered)-1\nGBR_pre = GBR_pre_casual+GBR_pre_registered\n\nsubmit1 = pd.DataFrame({'datetime':X_date,'count':0.2*RFR_pre+0.8*GBR_pre})\nsubmit1.to_csv('/kaggle/working/submisssion_1.csv',index=False)","metadata":{"ExecuteTime":{"end_time":"2022-05-25T13:11:38.079064Z","start_time":"2022-05-25T13:11:38.079064Z"},"execution":{"iopub.status.busy":"2022-06-04T09:07:15.882777Z","iopub.execute_input":"2022-06-04T09:07:15.883684Z","iopub.status.idle":"2022-06-04T09:07:17.066822Z","shell.execute_reply.started":"2022-06-04T09:07:15.883638Z","shell.execute_reply":"2022-06-04T09:07:17.065871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RFR_pre_count = RFR3.predict(RFR_X_test)\nRFR_pre_count = np.exp(RFR_pre_count)-1\nGBR_pre_count = GBR3.predict(GBR_X_test)\nGBR_pre_count = np.exp(GBR_pre_count)-1\n\npre_count=0.2*RFR_pre_count+0.8*GBR_pre_count\nsubmit2 = pd.DataFrame({'datetime':X_date,'count':pre_count})\nsubmit2.to_csv('/kaggle/working/submisssion_2.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-06-04T09:07:17.068006Z","iopub.execute_input":"2022-06-04T09:07:17.068282Z","iopub.status.idle":"2022-06-04T09:07:17.727081Z","shell.execute_reply.started":"2022-06-04T09:07:17.068256Z","shell.execute_reply":"2022-06-04T09:07:17.725991Z"},"trusted":true},"execution_count":null,"outputs":[]}]}