{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nfrom datetime import datetime\nimport warnings\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom lightgbm import LGBMRegressor\nimport calendar","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-25T03:09:44.115198Z","iopub.execute_input":"2022-07-25T03:09:44.115621Z","iopub.status.idle":"2022-07-25T03:09:44.123175Z","shell.execute_reply.started":"2022-07-25T03:09:44.115581Z","shell.execute_reply":"2022-07-25T03:09:44.122275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#去掉烦人的警告\nwarnings.filterwarnings('ignore')\n#设置绘图风格\nsns.set(style='whitegrid' , palette='deep')","metadata":{"execution":{"iopub.status.busy":"2022-07-25T03:09:44.371115Z","iopub.execute_input":"2022-07-25T03:09:44.372419Z","iopub.status.idle":"2022-07-25T03:09:44.377964Z","shell.execute_reply.started":"2022-07-25T03:09:44.372374Z","shell.execute_reply":"2022-07-25T03:09:44.377105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#导入训练数据\ntrain=pd.read_csv('../input/bike-sharing-demand/train.csv')\n#导入测试数据\ntest=pd.read_csv('../input/bike-sharing-demand/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-25T03:09:44.380781Z","iopub.execute_input":"2022-07-25T03:09:44.381430Z","iopub.status.idle":"2022-07-25T03:09:44.423345Z","shell.execute_reply.started":"2022-07-25T03:09:44.381388Z","shell.execute_reply":"2022-07-25T03:09:44.422406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#查看数据前五行\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T03:09:44.425001Z","iopub.execute_input":"2022-07-25T03:09:44.425388Z","iopub.status.idle":"2022-07-25T03:09:44.442523Z","shell.execute_reply.started":"2022-07-25T03:09:44.425355Z","shell.execute_reply":"2022-07-25T03:09:44.441328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T03:09:44.443927Z","iopub.execute_input":"2022-07-25T03:09:44.445050Z","iopub.status.idle":"2022-07-25T03:09:44.465731Z","shell.execute_reply.started":"2022-07-25T03:09:44.445010Z","shell.execute_reply":"2022-07-25T03:09:44.464408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#查看数据形状\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-25T03:09:44.469482Z","iopub.execute_input":"2022-07-25T03:09:44.470610Z","iopub.status.idle":"2022-07-25T03:09:44.480024Z","shell.execute_reply.started":"2022-07-25T03:09:44.470539Z","shell.execute_reply":"2022-07-25T03:09:44.478817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 查看训练集是否有缺失值\ntrain.info()\n#观察训练集数据描述统计\ntrain.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-07-25T03:09:44.481457Z","iopub.execute_input":"2022-07-25T03:09:44.482030Z","iopub.status.idle":"2022-07-25T03:09:44.551103Z","shell.execute_reply.started":"2022-07-25T03:09:44.481983Z","shell.execute_reply":"2022-07-25T03:09:44.550247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 绘制租赁额密度分布图\nsns.distplot(train['count'])","metadata":{"execution":{"iopub.status.busy":"2022-07-25T03:09:44.552221Z","iopub.execute_input":"2022-07-25T03:09:44.553081Z","iopub.status.idle":"2022-07-25T03:09:44.954060Z","shell.execute_reply.started":"2022-07-25T03:09:44.553045Z","shell.execute_reply":"2022-07-25T03:09:44.952696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 去除与租赁租赁数量平均值相差3个标准差的租赁数据\noutliers = np.abs(train['count']-train['count'].mean()) > (3*train['count'].std())\noutliers_num = len(train[outliers])\ntrain.drop(index=train[outliers].index)\nprint(\"一共去除了\",outliers_num,\"个租赁数据\")","metadata":{"execution":{"iopub.status.busy":"2022-07-25T03:09:44.957310Z","iopub.execute_input":"2022-07-25T03:09:44.957798Z","iopub.status.idle":"2022-07-25T03:09:44.970781Z","shell.execute_reply.started":"2022-07-25T03:09:44.957763Z","shell.execute_reply":"2022-07-25T03:09:44.969446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#绘制去除后的数据密度分布\nsns.distplot(train['count'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T03:09:44.972147Z","iopub.execute_input":"2022-07-25T03:09:44.973002Z","iopub.status.idle":"2022-07-25T03:09:45.405645Z","shell.execute_reply.started":"2022-07-25T03:09:44.972941Z","shell.execute_reply":"2022-07-25T03:09:45.404290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#对时间进行提取\ndef time_process(df):\n    #年、月、日、小时特征提取\n    df['year'] = pd.DatetimeIndex(df['datetime']).year\n    df['month'] = pd.DatetimeIndex(df['datetime']).month\n    df['day'] = pd.DatetimeIndex(df['datetime']).day\n    df['hour'] = pd.DatetimeIndex(df['datetime']).hour\n    #将日期的礼拜数标出，以探究工作日、双休日的特征\n    df['week'] = pd.DatetimeIndex(df['datetime']).weekofyear\n    df['weekday'] = pd.DatetimeIndex(df['datetime']).dayofweek\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-25T03:09:45.407319Z","iopub.execute_input":"2022-07-25T03:09:45.407816Z","iopub.status.idle":"2022-07-25T03:09:45.416394Z","shell.execute_reply.started":"2022-07-25T03:09:45.407771Z","shell.execute_reply":"2022-07-25T03:09:45.415145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#提取时间特征\ntrain = time_process(train)\ntest = time_process(test)\n\ncount=train['count'].values\ndate=test['datetime'].values","metadata":{"execution":{"iopub.status.busy":"2022-07-25T03:09:45.418664Z","iopub.execute_input":"2022-07-25T03:09:45.419213Z","iopub.status.idle":"2022-07-25T03:09:45.491405Z","shell.execute_reply.started":"2022-07-25T03:09:45.419094Z","shell.execute_reply":"2022-07-25T03:09:45.490340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#对风速的零值进行随机森林填充，基于季节和气温\ndef wind_fill(df):\n    wind_0 = df[df['windspeed']==0]\n    wind_not0 = df[df['windspeed']!=0]\n    y_label = wind_not0['windspeed']\n    #猜测风速和天气以及时间都有关\n    clf = RandomForestClassifier(n_estimators=1000,max_depth=10,random_state=0)\n    windcolunms = ['season', 'weather', 'temp', 'atemp', 'humidity', 'hour', 'month']\n    clf.fit(wind_not0[windcolunms], y_label.astype('int'))\n    pred_y = clf.predict(wind_0[windcolunms])\n    #预测结果填充\n    wind_0['windspeed'] = pred_y\n    df_rfw = wind_not0.append(wind_0)\n    df_rfw.reset_index(inplace=True)\n    return df_rfw\n\ntrain = wind_fill(train)\ntest = wind_fill(test)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T03:09:45.492851Z","iopub.execute_input":"2022-07-25T03:09:45.493290Z","iopub.status.idle":"2022-07-25T03:09:59.183593Z","shell.execute_reply.started":"2022-07-25T03:09:45.493257Z","shell.execute_reply":"2022-07-25T03:09:59.182257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#可视化分析\n\n#不同时间段--使用量\nsns.boxplot(x='hour',y='count',data=train)\n#根据数据得出使用量高峰时段\ntrain['peak'] = train[['hour', 'workingday']].apply(lambda x: (0, 1)[(x['workingday'] == 1 and  ( x['hour'] == 8 or 17 <= x['hour'] <= 18 or 12 <= x['hour'] <= 12)) or (x['workingday'] == 0 and  10 <= x['hour'] <= 19)], axis = 1)\ntest['peak'] = test[['hour', 'workingday']].apply(lambda x: (0, 1)[(x['workingday'] == 1 and  ( x['hour'] == 8 or 17 <= x['hour'] <= 18 or 12 <= x['hour'] <= 12)) or (x['workingday'] == 0 and  10 <= x['hour'] <= 19)], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T03:09:59.185246Z","iopub.execute_input":"2022-07-25T03:09:59.185626Z","iopub.status.idle":"2022-07-25T03:10:00.401529Z","shell.execute_reply.started":"2022-07-25T03:09:59.185584Z","shell.execute_reply":"2022-07-25T03:10:00.400349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 对数据count进行转换成其对数形式\n# yLabels=train['count']\n# yLabels_log=np.log(yLabels)\n# sns.distplot(yLabels_log)\n\n# yLabels1=train['casual']\n# yLabels1_log=np.log(yLabels1)\n# # sns.distplot(yLabels1_log)\n\n# yLabels2=train['registered']\n# yLabels2_log=np.log(yLabels2)\n# # sns.distplot(yLabels2_log)\n\ny_casual=train['casual'].apply(lambda x: np.log1p(x)).values\ny_registered=train['registered'].apply(lambda x: np.log1p(x)).values\ny_count=train['count'].apply(lambda x: np.log1p(x)).values\n","metadata":{"execution":{"iopub.status.busy":"2022-07-25T03:10:00.405052Z","iopub.execute_input":"2022-07-25T03:10:00.405437Z","iopub.status.idle":"2022-07-25T03:10:00.494041Z","shell.execute_reply.started":"2022-07-25T03:10:00.405404Z","shell.execute_reply":"2022-07-25T03:10:00.492806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 将多类别型数据使用one-hot转化成多个二分型类别\ndummies_month = pd.get_dummies(train['month'], prefix= 'month')\ndummies_season=pd.get_dummies(train['season'],prefix='season')\ndummies_weather=pd.get_dummies(train['weather'],prefix='weather')\ndummies_year=pd.get_dummies(train['year'],prefix='year')\n#把5个新的DF和原来的表连接起来\ntrain=pd.concat([train,dummies_month,dummies_season,dummies_weather,dummies_year],axis=1)\n\n\n# 将多类别型数据使用one-hot转化成多个二分型类别\ndummies_month = pd.get_dummies(test['month'], prefix= 'month')\ndummies_season=pd.get_dummies(test['season'],prefix='season')\ndummies_weather=pd.get_dummies(test['weather'],prefix='weather')\ndummies_year=pd.get_dummies(test['year'],prefix='year')\n#把5个新的DF和原来的表连接起来\ntest=pd.concat([test,dummies_month,dummies_season,dummies_weather,dummies_year],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T03:10:00.495431Z","iopub.execute_input":"2022-07-25T03:10:00.496303Z","iopub.status.idle":"2022-07-25T03:10:00.518472Z","shell.execute_reply.started":"2022-07-25T03:10:00.496266Z","shell.execute_reply":"2022-07-25T03:10:00.517117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #删去部分列\n# dropfeatures0='datetime'\n# dropfeatures3=['count','registered','casual']\n\n# y_count=train['count'].apply(lambda x: np.log1p(x)).values\n\n# train=train.drop(dropfeatures0, axis=1)\n# train=train.drop(dropfeatures3, axis=1)\n# test=test.drop(dropfeatures0,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T03:10:00.519978Z","iopub.execute_input":"2022-07-25T03:10:00.520454Z","iopub.status.idle":"2022-07-25T03:10:00.525597Z","shell.execute_reply.started":"2022-07-25T03:10:00.520416Z","shell.execute_reply":"2022-07-25T03:10:00.524148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_names = ['season', 'holiday', 'workingday', 'weather',\n                 'temp', 'atemp', 'humidity', 'windspeed',\n                 \"year\", \"hour\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T03:10:00.527070Z","iopub.execute_input":"2022-07-25T03:10:00.528087Z","iopub.status.idle":"2022-07-25T03:10:00.538310Z","shell.execute_reply.started":"2022-07-25T03:10:00.528049Z","shell.execute_reply":"2022-07-25T03:10:00.537315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#随机森林模型\nparams = {'n_estimators': 1000, \n          'max_depth': 15, \n          'random_state': 0, \n          'min_samples_split' : 5, \n          'n_jobs': -1}\n\nRFR = RandomForestRegressor(**params)\nRFR.fit(train[feature_names],y_count)\nprint(\"model拟合程度:\",RFR.score(train[feature_names],y_count))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T03:10:00.539989Z","iopub.execute_input":"2022-07-25T03:10:00.540734Z","iopub.status.idle":"2022-07-25T03:10:11.608427Z","shell.execute_reply.started":"2022-07-25T03:10:00.540698Z","shell.execute_reply":"2022-07-25T03:10:11.607181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#集成学习梯度提升决策树\nparams2 = {'n_estimators': 150, \n           'max_depth': 5, \n           'random_state': 0, \n           'min_samples_leaf' : 10, \n           'learning_rate': 0.1, \n           'subsample': 0.7, \n           'loss': 'ls'}\n\nGBR = GradientBoostingRegressor(**params2)\nGBR.fit(train[feature_names],y_count)\nprint(\"model拟合程度:\",GBR.score(train[feature_names],y_count))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T03:10:11.610056Z","iopub.execute_input":"2022-07-25T03:10:11.610431Z","iopub.status.idle":"2022-07-25T03:10:13.264769Z","shell.execute_reply.started":"2022-07-25T03:10:11.610400Z","shell.execute_reply":"2022-07-25T03:10:13.263552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RFR_pre_count = RFR.predict(test[feature_names])\n# RFR_pre_count = np.exp(RFR_pre_count)\n\n# RFR_pre_casual = RFR1.predict(test[feature_names])\n# RFR_pre_registered = RFR1.predict(test[feature_names])\n# RFR_pre_count = RFR_pre_casual + RFR_pre_registered\n\nGBR_pre_count = GBR.predict(test[feature_names])\n# GBR_pre_count = np.exp(GBR_pre_count)\n\n# GBR_pre_casual = GBR1.predict(test[feature_names])\n# GBR_pre_registered = GBR1.predict(test[feature_names])\n# GBR_pre_count = GBR_pre_casual + GBR_pre_registered","metadata":{"execution":{"iopub.status.busy":"2022-07-25T03:10:13.266423Z","iopub.execute_input":"2022-07-25T03:10:13.266848Z","iopub.status.idle":"2022-07-25T03:10:13.800990Z","shell.execute_reply.started":"2022-07-25T03:10:13.266815Z","shell.execute_reply":"2022-07-25T03:10:13.799785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit = pd.DataFrame({'datetime':date,'count':[max(0,x) for x in np.exp(0.5*RFR_pre_count+0.5*GBR_pre_count)]})\nsubmit.to_csv('/kaggle/working/submisssion12345678.csv',index=False)\n\nsubmit1 = pd.DataFrame({'datetime':date,'count':[max(0,x) for x in np.exp(GBR_pre_count)]})\nsubmit1.to_csv('/kaggle/working/submisssion123456789.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T03:10:13.802494Z","iopub.execute_input":"2022-07-25T03:10:13.802867Z","iopub.status.idle":"2022-07-25T03:10:13.881935Z","shell.execute_reply.started":"2022-07-25T03:10:13.802834Z","shell.execute_reply":"2022-07-25T03:10:13.880561Z"},"trusted":true},"execution_count":null,"outputs":[]}]}