{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-13T08:11:40.059902Z","iopub.execute_input":"2022-07-13T08:11:40.060292Z","iopub.status.idle":"2022-07-13T08:11:40.095574Z","shell.execute_reply.started":"2022-07-13T08:11:40.060207Z","shell.execute_reply":"2022-07-13T08:11:40.094536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#读取数据并展示基本信息\n\ntrain=pd.read_csv(\"/kaggle/input/bike-sharing-demand/train.csv\")\ntest=pd.read_csv(\"/kaggle/input/bike-sharing-demand/test.csv\")\ntrain.info()\ntest.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T08:11:40.097669Z","iopub.execute_input":"2022-07-13T08:11:40.098095Z","iopub.status.idle":"2022-07-13T08:11:40.182197Z","shell.execute_reply.started":"2022-07-13T08:11:40.098057Z","shell.execute_reply":"2022-07-13T08:11:40.181302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T08:11:40.183607Z","iopub.execute_input":"2022-07-13T08:11:40.184018Z","iopub.status.idle":"2022-07-13T08:11:40.204321Z","shell.execute_reply.started":"2022-07-13T08:11:40.183976Z","shell.execute_reply":"2022-07-13T08:11:40.203248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#获取数据[count]的均值标准差等信息\ntrain['count'].describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T08:11:40.205621Z","iopub.execute_input":"2022-07-13T08:11:40.206552Z","iopub.status.idle":"2022-07-13T08:11:40.219541Z","shell.execute_reply.started":"2022-07-13T08:11:40.206517Z","shell.execute_reply":"2022-07-13T08:11:40.218348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n#租赁数据分布\nfig, axis = plt.subplots(1, 2)\nfig.set_size_inches(12, 5)\n\nsns.boxplot(data=train,y=\"count\",orient=\"v\",ax=axis[0])\nsns.distplot(train['count'],ax=axis[1])\n\naxis[0].set(ylabel=\"count\", title=\"box plot of count\")\naxis[1].set(xlabel=\"count\", title=\"distribution of count\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T08:11:40.222449Z","iopub.execute_input":"2022-07-13T08:11:40.223499Z","iopub.status.idle":"2022-07-13T08:11:41.207657Z","shell.execute_reply.started":"2022-07-13T08:11:40.223454Z","shell.execute_reply":"2022-07-13T08:11:41.206761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#删除超出3倍标准差的数据，共147个\ntrain = train.loc[np.abs(train['count']-train['count'].mean()) < (3*train['count'].std())]\ntrain.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T08:11:41.208944Z","iopub.execute_input":"2022-07-13T08:11:41.209411Z","iopub.status.idle":"2022-07-13T08:11:41.227674Z","shell.execute_reply.started":"2022-07-13T08:11:41.209364Z","shell.execute_reply":"2022-07-13T08:11:41.226732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#将数据进行处理，把年、月、日、时间作为特征加入数据集，共得到12个特征\nalldata=[train,test];\nfor i in alldata:\n    i['datetime']=pd.to_datetime(i[\"datetime\"])\n    \nfor a in alldata:\n    a[\"year\"]=[i.year for i in a[\"datetime\"]]\n    a[\"month\"]=[i.month for i in a[\"datetime\"]]\n    a[\"day\"]=[i.day for i in a[\"datetime\"]]\n    a[\"hour\"]=[i.hour for i in a[\"datetime\"]]","metadata":{"execution":{"iopub.status.busy":"2022-07-13T08:11:41.229228Z","iopub.execute_input":"2022-07-13T08:11:41.229909Z","iopub.status.idle":"2022-07-13T08:11:41.351533Z","shell.execute_reply.started":"2022-07-13T08:11:41.229868Z","shell.execute_reply":"2022-07-13T08:11:41.350748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T08:11:41.352625Z","iopub.execute_input":"2022-07-13T08:11:41.352967Z","iopub.status.idle":"2022-07-13T08:11:41.371576Z","shell.execute_reply.started":"2022-07-13T08:11:41.352932Z","shell.execute_reply":"2022-07-13T08:11:41.370670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#各个特征的可视化处理\nname_col=['season', 'holiday', 'workingday', 'weather', 'temp',\n             'atemp', 'humidity', 'windspeed','year','month','day','hour']\nfor i in range(12):\n    fig, axis = plt.subplots(1, 1)\n    fig.set_size_inches(12, 10)\n\n    sns.boxplot(data=train,y='count',x=name_col[i],orient=\"v\")\n    axis.set(ylabel=\"count\", title=\"box plot of \"+name_col[i])","metadata":{"execution":{"iopub.status.busy":"2022-07-13T08:11:41.372919Z","iopub.execute_input":"2022-07-13T08:11:41.373500Z","iopub.status.idle":"2022-07-13T08:11:48.244111Z","shell.execute_reply.started":"2022-07-13T08:11:41.373455Z","shell.execute_reply":"2022-07-13T08:11:48.243156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#查看各组数据与count的相关性 计算12个特征与count之间的相关系数\ncorrmat = train[['season', 'holiday', 'workingday', 'weather', 'temp',\n             'atemp', 'humidity', 'windspeed','year','month','day','hour','count']].corr()\n#corr()求相关系数矩阵\nfig,ax=plt.subplots(1,1)\nfig.set_size_inches(20,10)\nsns.heatmap(corrmat,cmap=\"RdBu\",square=True,annot=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T08:11:48.245456Z","iopub.execute_input":"2022-07-13T08:11:48.245916Z","iopub.status.idle":"2022-07-13T08:11:49.124615Z","shell.execute_reply.started":"2022-07-13T08:11:48.245877Z","shell.execute_reply":"2022-07-13T08:11:49.123596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#搭建ANN模型\nfrom sklearn.preprocessing import MinMaxScaler\nimport tensorflow.keras\nfrom keras.layers import Dense,Dropout\nfrom keras.models import Sequential\nfrom sklearn.model_selection import train_test_split\n\ndef ANN_model_for_regression():\n    model=Sequential()\n    model.add(Dense(48, input_dim=12, activation='relu')) \n    model.add(Dense(96, input_dim=48, activation='relu'))  \n    model.add(Dense(96, input_dim=96, activation='relu'))  \n    model.add(Dense(48, input_dim=96, activation='relu'))\n    model.add(Dense(1, kernel_initializer='normal'))  \n    \n    model.compile(loss='mean_squared_error', optimizer='adam')\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-07-13T08:11:49.125687Z","iopub.execute_input":"2022-07-13T08:11:49.126038Z","iopub.status.idle":"2022-07-13T08:11:55.375762Z","shell.execute_reply.started":"2022-07-13T08:11:49.126004Z","shell.execute_reply":"2022-07-13T08:11:55.374700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#获取训练集和测试集\ny_train=train[[\"count\"]]\ntrain.drop([\"datetime\",\"casual\",\"registered\",\"count\"],axis=1,inplace=True)\nx_train=train\nx_test=test.drop([\"datetime\"],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T08:11:55.377049Z","iopub.execute_input":"2022-07-13T08:11:55.377756Z","iopub.status.idle":"2022-07-13T08:11:55.390562Z","shell.execute_reply.started":"2022-07-13T08:11:55.377716Z","shell.execute_reply":"2022-07-13T08:11:55.388782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 训练模型\nmodel = ANN_model_for_regression()  \nhistory = model.fit(x_train, y_train, epochs=100)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T08:11:55.392683Z","iopub.execute_input":"2022-07-13T08:11:55.393065Z","iopub.status.idle":"2022-07-13T08:13:21.140659Z","shell.execute_reply.started":"2022-07-13T08:11:55.393030Z","shell.execute_reply":"2022-07-13T08:13:21.139644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred=model.predict(x_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T08:13:21.144911Z","iopub.execute_input":"2022-07-13T08:13:21.145233Z","iopub.status.idle":"2022-07-13T08:13:21.428524Z","shell.execute_reply.started":"2022-07-13T08:13:21.145204Z","shell.execute_reply":"2022-07-13T08:13:21.427606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred","metadata":{"execution":{"iopub.status.busy":"2022-07-13T08:13:21.429755Z","iopub.execute_input":"2022-07-13T08:13:21.430124Z","iopub.status.idle":"2022-07-13T08:13:21.438878Z","shell.execute_reply.started":"2022-07-13T08:13:21.430087Z","shell.execute_reply":"2022-07-13T08:13:21.437811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[\"count\"]=y_pred\ntest.reset_index(inplace=True)\ntest.loc[test[\"count\"]<=0,\"count\"]=0  # 将小于0的无效预测值改为0\ntest[['datetime','count']].to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T08:13:21.440260Z","iopub.execute_input":"2022-07-13T08:13:21.440745Z","iopub.status.idle":"2022-07-13T08:13:21.483473Z","shell.execute_reply.started":"2022-07-13T08:13:21.440708Z","shell.execute_reply":"2022-07-13T08:13:21.482485Z"},"trusted":true},"execution_count":null,"outputs":[]}]}