{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#本文参考了https://www.kaggle.com/code/bobby2014king/bike-sharing-demand-byc和https://blog.csdn.net/coffeetogether/article/details/118560093","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:46:06.995430Z","iopub.execute_input":"2022-07-16T07:46:06.996554Z","iopub.status.idle":"2022-07-16T07:46:07.001459Z","shell.execute_reply.started":"2022-07-16T07:46:06.996517Z","shell.execute_reply":"2022-07-16T07:46:07.000280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nfrom datetime import datetime\nimport warnings\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom lightgbm import LGBMRegressor\nimport calendar","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-16T07:46:07.003746Z","iopub.execute_input":"2022-07-16T07:46:07.004881Z","iopub.status.idle":"2022-07-16T07:46:07.018592Z","shell.execute_reply.started":"2022-07-16T07:46:07.004833Z","shell.execute_reply":"2022-07-16T07:46:07.017536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#去掉烦人的警告\nwarnings.filterwarnings('ignore')\n#设置绘图风格\nsns.set(style='whitegrid' , palette='deep')","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:46:07.021065Z","iopub.execute_input":"2022-07-16T07:46:07.022517Z","iopub.status.idle":"2022-07-16T07:46:07.033429Z","shell.execute_reply.started":"2022-07-16T07:46:07.022463Z","shell.execute_reply":"2022-07-16T07:46:07.032429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#导入训练数据\ntrain=pd.read_csv('../input/bike-sharing-demand/train.csv')\n#导入测试数据\ntest=pd.read_csv('../input/bike-sharing-demand/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:47:00.408751Z","iopub.execute_input":"2022-07-16T07:47:00.409326Z","iopub.status.idle":"2022-07-16T07:47:00.449496Z","shell.execute_reply.started":"2022-07-16T07:47:00.409291Z","shell.execute_reply":"2022-07-16T07:47:00.448195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#查看数据前五行\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:47:02.804332Z","iopub.execute_input":"2022-07-16T07:47:02.804743Z","iopub.status.idle":"2022-07-16T07:47:02.827086Z","shell.execute_reply.started":"2022-07-16T07:47:02.804707Z","shell.execute_reply":"2022-07-16T07:47:02.825561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:47:05.080431Z","iopub.execute_input":"2022-07-16T07:47:05.081002Z","iopub.status.idle":"2022-07-16T07:47:05.093846Z","shell.execute_reply.started":"2022-07-16T07:47:05.080961Z","shell.execute_reply":"2022-07-16T07:47:05.093017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#查看数据形状\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:47:06.969655Z","iopub.execute_input":"2022-07-16T07:47:06.970330Z","iopub.status.idle":"2022-07-16T07:47:06.976499Z","shell.execute_reply.started":"2022-07-16T07:47:06.970291Z","shell.execute_reply":"2022-07-16T07:47:06.975480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 查看训练集是否有缺失值\ntrain.info()\n#观察训练集数据描述统计\ntrain.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:47:11.367271Z","iopub.execute_input":"2022-07-16T07:47:11.367674Z","iopub.status.idle":"2022-07-16T07:47:11.433991Z","shell.execute_reply.started":"2022-07-16T07:47:11.367639Z","shell.execute_reply":"2022-07-16T07:47:11.432854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 查看训练集是否有缺失值\ntest.info()\n#观察训练集数据描述统计\ntest.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:47:14.103107Z","iopub.execute_input":"2022-07-16T07:47:14.103939Z","iopub.status.idle":"2022-07-16T07:47:14.145007Z","shell.execute_reply.started":"2022-07-16T07:47:14.103895Z","shell.execute_reply":"2022-07-16T07:47:14.144219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 绘制租赁额密度分布图\nsns.distplot(train['count'])","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:47:17.871993Z","iopub.execute_input":"2022-07-16T07:47:17.872912Z","iopub.status.idle":"2022-07-16T07:47:18.294726Z","shell.execute_reply.started":"2022-07-16T07:47:17.872877Z","shell.execute_reply":"2022-07-16T07:47:18.293637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 去除与租赁租赁数量平均值相差3个标准差的租赁数据\noutliers = np.abs(train['count']-train['count'].mean()) > (3*train['count'].std())\noutliers_num = len(train[outliers])\ntrain.drop(index=train[outliers].index)\nprint(\"一共去除了\",outliers_num,\"个租赁数据\")","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:47:25.154207Z","iopub.execute_input":"2022-07-16T07:47:25.154593Z","iopub.status.idle":"2022-07-16T07:47:25.165743Z","shell.execute_reply.started":"2022-07-16T07:47:25.154545Z","shell.execute_reply":"2022-07-16T07:47:25.164525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#绘制去除后的数据密度分布\nsns.distplot(train['count'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:47:28.016199Z","iopub.execute_input":"2022-07-16T07:47:28.016907Z","iopub.status.idle":"2022-07-16T07:47:28.377534Z","shell.execute_reply.started":"2022-07-16T07:47:28.016870Z","shell.execute_reply":"2022-07-16T07:47:28.376658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#对时间进行提取\ndef time_process(df):\n    #年、月、日、小时特征提取\n    df['year'] = pd.DatetimeIndex(df['datetime']).year\n    df['month'] = pd.DatetimeIndex(df['datetime']).month\n    df['day'] = pd.DatetimeIndex(df['datetime']).day\n    df['hour'] = pd.DatetimeIndex(df['datetime']).hour\n    #将日期的礼拜数标出，以探究工作日、双休日的特征\n    df['week'] = pd.DatetimeIndex(df['datetime']).weekofyear\n    df['weekday'] = pd.DatetimeIndex(df['datetime']).dayofweek\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:47:31.131888Z","iopub.execute_input":"2022-07-16T07:47:31.133030Z","iopub.status.idle":"2022-07-16T07:47:31.138858Z","shell.execute_reply.started":"2022-07-16T07:47:31.132986Z","shell.execute_reply":"2022-07-16T07:47:31.137839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#提取时间特征\ntrain = time_process(train)\ntest = time_process(test)\n\ncount=train['count'].values\ndate=test['datetime'].values","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:47:42.233493Z","iopub.execute_input":"2022-07-16T07:47:42.233878Z","iopub.status.idle":"2022-07-16T07:47:42.282564Z","shell.execute_reply.started":"2022-07-16T07:47:42.233848Z","shell.execute_reply":"2022-07-16T07:47:42.281487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#对风速的零值进行随机森林填充，基于季节和气温\ndef wind_fill(df):\n    wind_0 = df[df['windspeed']==0]\n    wind_not0 = df[df['windspeed']!=0]\n    y_label = wind_not0['windspeed']\n    #猜测风速和天气以及时间都有关\n    clf = RandomForestClassifier(n_estimators=1000,max_depth=10,random_state=0)\n    windcolunms = ['season', 'weather', 'temp', 'atemp', 'humidity', 'hour', 'month']\n    clf.fit(wind_not0[windcolunms], y_label.astype('int'))\n    pred_y = clf.predict(wind_0[windcolunms])\n    #预测结果填充\n    wind_0['windspeed'] = pred_y\n    df_rfw = wind_not0.append(wind_0)\n    df_rfw.reset_index(inplace=True)\n    return df_rfw\n\ntrain = wind_fill(train)\ntest = wind_fill(test)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:47:43.782192Z","iopub.execute_input":"2022-07-16T07:47:43.783176Z","iopub.status.idle":"2022-07-16T07:47:56.145486Z","shell.execute_reply.started":"2022-07-16T07:47:43.783132Z","shell.execute_reply":"2022-07-16T07:47:56.144295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#可视化分析\n\n#不同时间段--使用量\nsns.boxplot(x='hour',y='count',data=train)\n#根据数据得出使用量高峰时段\ntrain['peak'] = train[['hour', 'workingday']].apply(lambda x: (0, 1)[(x['workingday'] == 1 and  ( x['hour'] == 8 or 17 <= x['hour'] <= 18 or 12 <= x['hour'] <= 12)) or (x['workingday'] == 0 and  10 <= x['hour'] <= 19)], axis = 1)\ntest['peak'] = test[['hour', 'workingday']].apply(lambda x: (0, 1)[(x['workingday'] == 1 and  ( x['hour'] == 8 or 17 <= x['hour'] <= 18 or 12 <= x['hour'] <= 12)) or (x['workingday'] == 0 and  10 <= x['hour'] <= 19)], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:47:56.147853Z","iopub.execute_input":"2022-07-16T07:47:56.148274Z","iopub.status.idle":"2022-07-16T07:47:57.149760Z","shell.execute_reply.started":"2022-07-16T07:47:56.148229Z","shell.execute_reply":"2022-07-16T07:47:57.148700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 对数据count进行转换成其对数形式\n# yLabels=train['count']\n# yLabels_log=np.log(yLabels)\n# sns.distplot(yLabels_log)\n\n# yLabels1=train['casual']\n# yLabels1_log=np.log(yLabels1)\n# # sns.distplot(yLabels1_log)\n\n# yLabels2=train['registered']\n# yLabels2_log=np.log(yLabels2)\n# # sns.distplot(yLabels2_log)\n\ny_casual=train['casual'].apply(lambda x: np.log1p(x)).values\ny_registered=train['registered'].apply(lambda x: np.log1p(x)).values\ny_count=train['count'].apply(lambda x: np.log1p(x)).values\n","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:48:00.235536Z","iopub.execute_input":"2022-07-16T07:48:00.235915Z","iopub.status.idle":"2022-07-16T07:48:00.291823Z","shell.execute_reply.started":"2022-07-16T07:48:00.235884Z","shell.execute_reply":"2022-07-16T07:48:00.290729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 将多类别型数据使用one-hot转化成多个二分型类别\ndummies_month = pd.get_dummies(train['month'], prefix= 'month')\ndummies_season=pd.get_dummies(train['season'],prefix='season')\ndummies_weather=pd.get_dummies(train['weather'],prefix='weather')\ndummies_year=pd.get_dummies(train['year'],prefix='year')\n#把5个新的DF和原来的表连接起来\ntrain=pd.concat([train,dummies_month,dummies_season,dummies_weather,dummies_year],axis=1)\n\n\n# 将多类别型数据使用one-hot转化成多个二分型类别\ndummies_month = pd.get_dummies(test['month'], prefix= 'month')\ndummies_season=pd.get_dummies(test['season'],prefix='season')\ndummies_weather=pd.get_dummies(test['weather'],prefix='weather')\ndummies_year=pd.get_dummies(test['year'],prefix='year')\n#把5个新的DF和原来的表连接起来\ntest=pd.concat([test,dummies_month,dummies_season,dummies_weather,dummies_year],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:48:06.697987Z","iopub.execute_input":"2022-07-16T07:48:06.698357Z","iopub.status.idle":"2022-07-16T07:48:06.719337Z","shell.execute_reply.started":"2022-07-16T07:48:06.698324Z","shell.execute_reply":"2022-07-16T07:48:06.718225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #删去部分列\n# dropfeatures0='datetime'\n# dropfeatures3=['count','registered','casual']\n\n# y_count=train['count'].apply(lambda x: np.log1p(x)).values\n\n# train=train.drop(dropfeatures0, axis=1)\n# train=train.drop(dropfeatures3, axis=1)\n# test=test.drop(dropfeatures0,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:46:07.284353Z","iopub.status.idle":"2022-07-16T07:46:07.284942Z","shell.execute_reply.started":"2022-07-16T07:46:07.284755Z","shell.execute_reply":"2022-07-16T07:46:07.284774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_names = ['season', 'holiday', 'workingday', 'weather',\n                 'temp', 'atemp', 'humidity', 'windspeed',\n                 \"year\", \"hour\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:48:24.790795Z","iopub.execute_input":"2022-07-16T07:48:24.791167Z","iopub.status.idle":"2022-07-16T07:48:24.796600Z","shell.execute_reply.started":"2022-07-16T07:48:24.791139Z","shell.execute_reply":"2022-07-16T07:48:24.795118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#随机森林模型\nparams = {'n_estimators': 1000, \n          'max_depth': 15, \n          'random_state': 0, \n          'min_samples_split' : 5, \n          'n_jobs': -1}\n\nRFR = RandomForestRegressor(**params)\nRFR.fit(train[feature_names],y_count)\nprint(\"model拟合程度:\",RFR.score(train[feature_names],y_count))","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:48:43.886472Z","iopub.execute_input":"2022-07-16T07:48:43.886859Z","iopub.status.idle":"2022-07-16T07:48:54.534052Z","shell.execute_reply.started":"2022-07-16T07:48:43.886828Z","shell.execute_reply":"2022-07-16T07:48:54.532871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#集成学习梯度提升决策树\nparams2 = {'n_estimators': 150, \n           'max_depth': 5, \n           'random_state': 0, \n           'min_samples_leaf' : 10, \n           'learning_rate': 0.1, \n           'subsample': 0.7, \n           'loss': 'ls'}\n\nGBR = GradientBoostingRegressor(**params2)\nGBR.fit(train[feature_names],y_count)\nprint(\"model拟合程度:\",GBR.score(train[feature_names],y_count))","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:48:28.431651Z","iopub.execute_input":"2022-07-16T07:48:28.432019Z","iopub.status.idle":"2022-07-16T07:48:30.044505Z","shell.execute_reply.started":"2022-07-16T07:48:28.431988Z","shell.execute_reply":"2022-07-16T07:48:30.042998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RFR_pre_count = RFR.predict(test[feature_names])\n# RFR_pre_count = np.exp(RFR_pre_count)\n\n# RFR_pre_casual = RFR1.predict(test[feature_names])\n# RFR_pre_registered = RFR1.predict(test[feature_names])\n# RFR_pre_count = RFR_pre_casual + RFR_pre_registered\n\nGBR_pre_count = GBR.predict(test[feature_names])\n# GBR_pre_count = np.exp(GBR_pre_count)\n\n# GBR_pre_casual = GBR1.predict(test[feature_names])\n# GBR_pre_registered = GBR1.predict(test[feature_names])\n# GBR_pre_count = GBR_pre_casual + GBR_pre_registered","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:50:09.479647Z","iopub.execute_input":"2022-07-16T07:50:09.480595Z","iopub.status.idle":"2022-07-16T07:50:10.013230Z","shell.execute_reply.started":"2022-07-16T07:50:09.480535Z","shell.execute_reply":"2022-07-16T07:50:10.012081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit = pd.DataFrame({'datetime':date,'count':[max(0,x) for x in np.exp(0.5*RFR_pre_count+0.5*GBR_pre_count)]})\nsubmit.to_csv('/kaggle/working/submisssion12345678.csv',index=False)\n\nsubmit1 = pd.DataFrame({'datetime':date,'count':[max(0,x) for x in np.exp(GBR_pre_count)]})\nsubmit1.to_csv('/kaggle/working/submisssion123456789.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:49:30.019552Z","iopub.execute_input":"2022-07-16T07:49:30.019995Z","iopub.status.idle":"2022-07-16T07:49:30.087461Z","shell.execute_reply.started":"2022-07-16T07:49:30.019949Z","shell.execute_reply":"2022-07-16T07:49:30.086295Z"},"trusted":true},"execution_count":null,"outputs":[]}]}