{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-15T15:31:36.368652Z","iopub.execute_input":"2022-07-15T15:31:36.369046Z","iopub.status.idle":"2022-07-15T15:31:36.382982Z","shell.execute_reply.started":"2022-07-15T15:31:36.369013Z","shell.execute_reply":"2022-07-15T15:31:36.382042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"我们本次的目标是预测自行车租赁数\n\n以下是提供的信息\n\ndatetime-每小时日期+时间戳\n\n季节-1=春季，2=夏季，3=秋季，4=冬季\n\n假日-这一天是否为假日\n\n工作日-无论这一天既不是周末也不是假日\n\n天气-1：晴朗，少云，部分多云，部分多云\n2： 雾+多云，雾+碎云，雾+少云，雾\n3： 小雪、小雨+雷雨+散云、小雨+散云\n4： 大雨+冰托盘+雷雨+雾，雪+雾\n（以上天气为机翻）\n\ntemp-温度，单位为摄氏度\n\natemp-体感温度，单位为摄氏度\n\n湿度-相对湿度\n\n风速-风速\n\ncasual非注册用户-非注册用户租用数量\n\nregistered注册用户-注册用户租用数量\n\ncount-总租借数","metadata":{}},{"cell_type":"markdown","source":"**查看相应的训练数据和测试数据**","metadata":{}},{"cell_type":"code","source":"#导入并查看训练数据和测试数据\ntrain_data = pd.read_csv('/kaggle/input/bike-sharing-demand/train.csv')\ntest_data = pd.read_csv('/kaggle/input/bike-sharing-demand/test.csv')\nprint(train_data.shape)\nprint(\"\\n\")\nprint(train_data.info())\nprint(\"\\n\")\nprint(test_data.shape)\nprint(\"\\n\")\nprint(test_data.info())\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:31:36.391391Z","iopub.execute_input":"2022-07-15T15:31:36.391668Z","iopub.status.idle":"2022-07-15T15:31:36.443151Z","shell.execute_reply.started":"2022-07-15T15:31:36.391642Z","shell.execute_reply":"2022-07-15T15:31:36.441953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"从上述我们可以看到了训练数据12列，10886行。测试集为9列，6493行。其中少了注册与非注册用户租借数量以及租借总量这三个属性。\n其中租借总量正是我们需要预测的。\n\n而且以上没有空数据，不需要我们填补","metadata":{}},{"cell_type":"code","source":"from datetime import datetime\n\n#第二步：数据预处理\n#合并两种数据，使之共同进行数据规范化\ndata = train_data.append(test_data)\n#拆分年、月、日、时\ndata['year'] = data.datetime.apply(lambda x: x.split()[0].split('-')[0])\ndata['year'] = data['year'].apply(lambda x: int(x))\ndata['month'] = data.datetime.apply(lambda x: x.split()[0].split('-')[1])\ndata['month'] = data['month'].apply(lambda x: int(x))\ndata['day'] = data.datetime.apply(lambda x: x.split()[0].split('-')[2])\ndata['day'] = data['day'].apply(lambda x: int(x))\ndata['hour'] = data.datetime.apply(lambda x: x.split()[1].split(':')[0])\ndata['hour'] = data['hour'].apply(lambda x: int(x))\ndata['date'] = data.datetime.apply(lambda x: x.split()[0])\ndata = data.drop('datetime',axis=1)\n#重新安排整体数据的特征\ncols = ['year','month','day','hour','season','holiday','workingday','weather','temp','atemp',\n        'humidity','windspeed','casual','registered','count']\ndata = data.loc[:,cols]\n#分离训练数据与测试数据\ntrain = data.iloc[:10886]\ntest = data.iloc[10886:]\nprint(data.shape)\nprint(data.info())","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:31:36.445830Z","iopub.execute_input":"2022-07-15T15:31:36.446228Z","iopub.status.idle":"2022-07-15T15:31:36.579793Z","shell.execute_reply.started":"2022-07-15T15:31:36.446189Z","shell.execute_reply":"2022-07-15T15:31:36.577761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"#第三步：特征工程\n#1、计算相关系数，并快速查看\ncorrelation = train.corr()\ninfluence_order = correlation['count'].sort_values(ascending=False)\ninfluence_order_abs = abs(correlation['count']).sort_values(ascending=False)\nprint(influence_order)\nprint(influence_order_abs)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:31:36.581467Z","iopub.execute_input":"2022-07-15T15:31:36.581883Z","iopub.status.idle":"2022-07-15T15:31:36.600216Z","shell.execute_reply.started":"2022-07-15T15:31:36.581841Z","shell.execute_reply":"2022-07-15T15:31:36.599089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#2、作相关性分析的热力图\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nf,ax = plt.subplots(figsize=(16,16))\ncmap = sn.cubehelix_palette(light=1,as_cmap=True)\nsn.heatmap(correlation,annot=True,center=1,cmap=cmap,linewidths=1,ax=ax)\nsn.heatmap(correlation,vmax=1,square=True,annot=True,linewidths=1)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:31:36.603061Z","iopub.execute_input":"2022-07-15T15:31:36.603432Z","iopub.status.idle":"2022-07-15T15:31:38.729468Z","shell.execute_reply.started":"2022-07-15T15:31:36.603386Z","shell.execute_reply":"2022-07-15T15:31:38.728552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#3、每个特征对租借量的影响\n#(1) 时间维度——年份\nsns.boxplot(x=train['year'],y=train['count'])\nplt.title(\"The influence of year\")\nplt.show()\n#(2) 时间维度——月份\nsns.pointplot(x=train['month'],y=train['count'])\nplt.title(\"The influence of month\")\nplt.show()\n#(3) 时间维度——季节\nsns.boxplot(x=train['season'],y=train['count'])\nplt.title(\"The influence of season\")\nplt.show()\n#(4) 时间维度——时间（小时）\nsns.barplot(x=train['hour'],y=train['count'])\nplt.title(\"The influence of hour\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:31:38.731191Z","iopub.execute_input":"2022-07-15T15:31:38.731787Z","iopub.status.idle":"2022-07-15T15:31:40.850755Z","shell.execute_reply.started":"2022-07-15T15:31:38.731752Z","shell.execute_reply":"2022-07-15T15:31:40.849813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#星期、节假日和工作日的影响\nfig, axes = plt.subplots(figsize=(16, 10))\nax1 = plt.subplot(2,2,1)\nsns.pointplot(x=train['hour'],y=train['count'],hue=train['workingday'],ax=ax1)\nax1.set_title(\"The influence of hour (workingday)\")\nax2 = plt.subplot(2,2,2)\nsns.pointplot(x=train['hour'],y=train['count'],hue=train['holiday'],ax=ax2)\nax2.set_title(\"The influence of hour (holiday)\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:31:40.852239Z","iopub.execute_input":"2022-07-15T15:31:40.852624Z","iopub.status.idle":"2022-07-15T15:31:44.745471Z","shell.execute_reply.started":"2022-07-15T15:31:40.852572Z","shell.execute_reply":"2022-07-15T15:31:44.744544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#(5) 天气的影响\nsns.boxplot(x=train['weather'],y=train['count'])\nplt.title(\"The influence of weather\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:31:44.747115Z","iopub.execute_input":"2022-07-15T15:31:44.748117Z","iopub.status.idle":"2022-07-15T15:31:44.949554Z","shell.execute_reply.started":"2022-07-15T15:31:44.748079Z","shell.execute_reply":"2022-07-15T15:31:44.948568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#(6) 温度、湿度、风速的影响\ncols = ['temp', 'atemp', 'humidity', 'windspeed', 'count']\nsns.pairplot(train[cols])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:31:44.952425Z","iopub.execute_input":"2022-07-15T15:31:44.953134Z","iopub.status.idle":"2022-07-15T15:31:50.554782Z","shell.execute_reply.started":"2022-07-15T15:31:44.953101Z","shell.execute_reply":"2022-07-15T15:31:50.553764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfig, axes = plt.subplots(1,3,figsize=(24,8))\nax1 = plt.subplot(1,3,1)\nax2 = plt.subplot(1,3,2)\nax3 = plt.subplot(1,3,3)\nsns.regplot(x=train['temp'],y=train['count'],ax=ax1)\nsns.regplot(x=train['humidity'],y=train['count'],ax=ax2)\nsns.regplot(x=train['windspeed'],y=train['count'],ax=ax3)\nax1.set_title(\"The influence of temperature\")\nax2.set_title(\"The influence of humidity\")\nax3.set_title(\"The influence of windspeed\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:31:50.556523Z","iopub.execute_input":"2022-07-15T15:31:50.557822Z","iopub.status.idle":"2022-07-15T15:31:53.215510Z","shell.execute_reply.started":"2022-07-15T15:31:50.557778Z","shell.execute_reply":"2022-07-15T15:31:53.214589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#特征工程\n#所选取的特征：year、month、hour、workingday、holiday、weather、temp、humidity和windspeed\n#(1) 删除不要的变量\ndata = data.drop(['day','season','atemp','casual','registered'],axis=1)\n#(2) 离散型变量（year、month、hour、weather）转换\ncolumn_trans = ['year','month','hour','weather']\ndata = pd.get_dummies(data, columns=column_trans)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:31:53.217292Z","iopub.execute_input":"2022-07-15T15:31:53.217847Z","iopub.status.idle":"2022-07-15T15:31:53.238432Z","shell.execute_reply.started":"2022-07-15T15:31:53.217813Z","shell.execute_reply":"2022-07-15T15:31:53.237561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")\n\n#机器学习\n#1、特征向量化\ncol_trans = ['holiday', 'workingday', 'temp', 'humidity', 'windspeed',\n       'year_2011', 'year_2012', 'month_1', 'month_2', 'month_3', 'month_4',\n       'month_5', 'month_6', 'month_7', 'month_8', 'month_9', 'month_10',\n       'month_11', 'month_12', 'hour_0', 'hour_1', 'hour_2', 'hour_3',\n       'hour_4', 'hour_5', 'hour_6', 'hour_7', 'hour_8', 'hour_9', 'hour_10',\n       'hour_11', 'hour_12', 'hour_13', 'hour_14', 'hour_15', 'hour_16',\n       'hour_17', 'hour_18', 'hour_19', 'hour_20', 'hour_21', 'hour_22',\n       'hour_23', 'weather_1', 'weather_2', 'weather_3', 'weather_4']\nX_train = data[col_trans].iloc[:10886]\nX_test = data[col_trans].iloc[10886:]\nY_train = data['count'].iloc[:10886]\nfrom sklearn.feature_extraction import DictVectorizer\nvec = DictVectorizer(sparse=False)\nX_train = vec.fit_transform(X_train.to_dict('record'))\nX_test = vec.fit_transform(X_test.to_dict('record'))\n \n#分割训练数据\nfrom sklearn.model_selection import train_test_split\nx_train, x_test, y_train, y_test = train_test_split(X_train, Y_train, test_size=0.25, random_state=40)\n \n#2、建模预测，分别采用常规集成学习方法、XGBoost和神经网络三大类模型\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.ensemble import ExtraTreesRegressor\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom xgboost import XGBRegressor\nfrom sklearn.neural_network import MLPRegressor\nfrom sklearn.metrics import r2_score\n \n#（1）集成学习方法——普通随机森林\nrfr = RandomForestRegressor()\nrfr.fit(x_train,y_train)\n#print(rfr.fit(x_train,y_train))\nrfr_y_predict = rfr.predict(x_test)\nprint(\"集成学习方法——普通随机森林回归模型的R方得分为：\",r2_score(y_test,rfr_y_predict))\n \n#（2）集成学习方法——极端随机森林\netr = ExtraTreesRegressor()\netr.fit(x_train,y_train)\n#print(etr.fit(x_train,y_train))\netr_y_predict = etr.predict(x_test)\nprint(\"集成学习方法——极端随机森林回归模型的R方得分为：\",r2_score(y_test,etr_y_predict))\n \n#（3）集成学习方法——梯度提升树\ngbr = GradientBoostingRegressor()\ngbr.fit(x_train,y_train)\n#print(gbr.fit(x_train,y_train))\ngbr_y_predict = gbr.predict(x_test)\nprint(\"集成学习方法——梯度提升树回归模型的R方得分为：\",r2_score(y_test,gbr_y_predict))\n \n#（4） XGBoost回归模型\nxgbr = XGBRegressor()\nxgbr.fit(x_train,y_train)\n#print(xgbr.fit(x_train,y_train))\nxgbr_y_predict = xgbr.predict(x_test)\nprint(\"XGBoost回归模型的R方得分为：\",r2_score(y_test,xgbr_y_predict))\n \n#（5） 神经网络回归模型\nmlp = MLPRegressor(hidden_layer_sizes=(47,47,47),max_iter=500)\nmlp.fit(x_train,y_train)\nmlp_y_predict = mlp.predict(x_test)\nprint(\"神经网络回归模型的R方得分为：\",r2_score(y_test,mlp_y_predict))\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:31:53.240100Z","iopub.execute_input":"2022-07-15T15:31:53.240454Z","iopub.status.idle":"2022-07-15T15:32:21.637257Z","shell.execute_reply.started":"2022-07-15T15:31:53.240418Z","shell.execute_reply":"2022-07-15T15:32:21.633673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"date=test_data['datetime'].values\n\nparams = {'n_estimators': 1000, \n          'max_depth': 38, \n          'random_state': 27, \n          'min_samples_split' : 2,\n          'n_jobs': -1}\n\nrfr = RandomForestRegressor(**params)\nrfr.fit(X_train,Y_train)\n\nrfr_predict = rfr.predict(X_test)\n\n# rf_pre_count = np.exp(rfr.predict(gb_x_test))+np.exp(rf_r.predict(rf_x_test))-2\n# xgb_pre_count = np.exp(xgb_c.predict(gb_x_test))+np.exp(xgb_r.predict(gb_x_test))-2\n\nxgbr = XGBRegressor(max_depth=20, learning_rate=0.1, random_state = 20,n_estimators=600)\nxgbr.fit(X_train,Y_train)\n\nxgbr_predict = xgbr.predict(X_test)\n\n\npre_count=np.round(0.2*rfr_predict+0.8*xgbr_predict,0)\npre_count=np.absolute(pre_count)\nsubmit = pd.DataFrame({'datetime':date,'count':pre_count})\nsubmit.to_csv('/kaggle/working/submisssion8.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:32:21.638865Z","iopub.execute_input":"2022-07-15T15:32:21.639212Z","iopub.status.idle":"2022-07-15T15:33:38.346976Z","shell.execute_reply.started":"2022-07-15T15:32:21.639175Z","shell.execute_reply":"2022-07-15T15:33:38.345928Z"},"trusted":true},"execution_count":null,"outputs":[]}]}