{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-12T12:48:04.647432Z","iopub.execute_input":"2022-07-12T12:48:04.647787Z","iopub.status.idle":"2022-07-12T12:48:04.663334Z","shell.execute_reply.started":"2022-07-12T12:48:04.647758Z","shell.execute_reply":"2022-07-12T12:48:04.662215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Exploration","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nfrom sklearn.metrics import mean_squared_error, mean_squared_log_error, r2_score\nfrom sklearn.linear_model import LinearRegression, Ridge\nfrom sklearn.model_selection import train_test_split, cross_validate\nfrom sklearn.ensemble import RandomForestRegressor\nfrom category_encoders import OneHotEncoder\nfrom sklearn.pipeline import make_pipeline","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:48:04.760765Z","iopub.execute_input":"2022-07-12T12:48:04.762255Z","iopub.status.idle":"2022-07-12T12:48:05.707881Z","shell.execute_reply.started":"2022-07-12T12:48:04.762210Z","shell.execute_reply":"2022-07-12T12:48:05.706936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"../input/bike-sharing-demand/train.csv\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:48:05.712137Z","iopub.execute_input":"2022-07-12T12:48:05.712423Z","iopub.status.idle":"2022-07-12T12:48:05.766599Z","shell.execute_reply.started":"2022-07-12T12:48:05.712400Z","shell.execute_reply":"2022-07-12T12:48:05.765714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"datetime\"] = pd.to_datetime(train[\"datetime\"])\n\ntrain[\"hour\"] = train[\"datetime\"].dt.hour\ntrain[\"day\"] = train[\"datetime\"].dt.day\ntrain[\"year\"] = train[\"datetime\"].dt.year\ntrain[\"month\"] = train[\"datetime\"].dt.month\n\ntrain[['datetime', 'day', 'month', 'year']].tail()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:48:05.769554Z","iopub.execute_input":"2022-07-12T12:48:05.769828Z","iopub.status.idle":"2022-07-12T12:48:05.809176Z","shell.execute_reply.started":"2022-07-12T12:48:05.769804Z","shell.execute_reply":"2022-07-12T12:48:05.808064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"day_of_the_week\"] = train[\"datetime\"].dt.dayofweek\ntrain[\"week_of_the_year\"] = train[\"datetime\"].dt.week\ntrain['is_leap_year'] = train['datetime'].dt.is_leap_year","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:48:42.441607Z","iopub.execute_input":"2022-07-12T12:48:42.441949Z","iopub.status.idle":"2022-07-12T12:48:42.458840Z","shell.execute_reply.started":"2022-07-12T12:48:42.441920Z","shell.execute_reply":"2022-07-12T12:48:42.457887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"week_mapping={\n    0: 'Monday', \n    1: 'Tuesday', \n    2: 'Wednesday', \n    3: 'Thursday', \n    4: 'Friday',\n    5: 'Saturday', \n    6: 'Sunday'\n} \ntrain['day_of_week_name']=train['datetime'].dt.weekday.map(week_mapping)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:48:46.039590Z","iopub.execute_input":"2022-07-12T12:48:46.040213Z","iopub.status.idle":"2022-07-12T12:48:46.049672Z","shell.execute_reply.started":"2022-07-12T12:48:46.040177Z","shell.execute_reply":"2022-07-12T12:48:46.048743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Time difference between the latest and oldest `datetime`","metadata":{}},{"cell_type":"code","source":"today = pd.to_datetime('today')\ntrain['period'] = today.year - train['datetime'].dt.year","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:48:53.189780Z","iopub.execute_input":"2022-07-12T12:48:53.190139Z","iopub.status.idle":"2022-07-12T12:48:53.196946Z","shell.execute_reply.started":"2022-07-12T12:48:53.190109Z","shell.execute_reply":"2022-07-12T12:48:53.196028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# shift the dates up and into a new column\ntrain['dates_shift'] = train['datetime'].shift(-1)\n\n# work out the diff\ntrain['time_diff'] = (train['dates_shift'] - train['datetime']) / pd.Timedelta(seconds=1)\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:48:56.582272Z","iopub.execute_input":"2022-07-12T12:48:56.582614Z","iopub.status.idle":"2022-07-12T12:48:56.611744Z","shell.execute_reply.started":"2022-07-12T12:48:56.582586Z","shell.execute_reply":"2022-07-12T12:48:56.610596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Also the highest and lowest time","metadata":{}},{"cell_type":"code","source":"train[\"datetime\"].max() - train[\"datetime\"].min()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:49:08.153724Z","iopub.execute_input":"2022-07-12T12:49:08.154078Z","iopub.status.idle":"2022-07-12T12:49:08.164598Z","shell.execute_reply.started":"2022-07-12T12:49:08.154049Z","shell.execute_reply":"2022-07-12T12:49:08.163649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:49:11.149784Z","iopub.execute_input":"2022-07-12T12:49:11.150153Z","iopub.status.idle":"2022-07-12T12:49:11.173272Z","shell.execute_reply.started":"2022-07-12T12:49:11.150123Z","shell.execute_reply":"2022-07-12T12:49:11.172225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\n\nimputer = SimpleImputer(missing_values=np.NaN, strategy='mean')\ntrain[\"dates_shift\"] = imputer.fit_transform(train['dates_shift'].values.reshape(-1,1))[:,0]\ntrain[\"time_diff\"] = imputer.fit_transform(train['time_diff'].values.reshape(-1,1))[:,0]\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:49:14.916356Z","iopub.execute_input":"2022-07-12T12:49:14.916726Z","iopub.status.idle":"2022-07-12T12:49:14.929481Z","shell.execute_reply.started":"2022-07-12T12:49:14.916699Z","shell.execute_reply":"2022-07-12T12:49:14.928468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:49:22.323456Z","iopub.execute_input":"2022-07-12T12:49:22.323831Z","iopub.status.idle":"2022-07-12T12:49:22.330559Z","shell.execute_reply.started":"2022-07-12T12:49:22.323783Z","shell.execute_reply":"2022-07-12T12:49:22.329560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:49:25.356747Z","iopub.execute_input":"2022-07-12T12:49:25.357253Z","iopub.status.idle":"2022-07-12T12:49:25.374365Z","shell.execute_reply.started":"2022-07-12T12:49:25.357208Z","shell.execute_reply":"2022-07-12T12:49:25.373437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(figsize=(12, 4))\ntrain.groupby(train[\"datetime\"].dt.hour)[\"temp\"].mean().plot(\n    kind='bar', rot=0, ax=axs)\nplt.xlabel(\"Hour of the day\")\nplt.ylabel(\"Temperature\");","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:49:28.957378Z","iopub.execute_input":"2022-07-12T12:49:28.958318Z","iopub.status.idle":"2022-07-12T12:49:29.248231Z","shell.execute_reply.started":"2022-07-12T12:49:28.958279Z","shell.execute_reply":"2022-07-12T12:49:29.247358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"season_temp = train.pivot(index=\"datetime\", columns=\"season\", values=\"temp\")\nseason_temp.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:49:32.696573Z","iopub.execute_input":"2022-07-12T12:49:32.697047Z","iopub.status.idle":"2022-07-12T12:49:32.725694Z","shell.execute_reply.started":"2022-07-12T12:49:32.696965Z","shell.execute_reply":"2022-07-12T12:49:32.724501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"season_temp[\"2011-01-01 00:00:00\":\"2012-12-19 23:00:00\"].plot();","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:49:36.689201Z","iopub.execute_input":"2022-07-12T12:49:36.689542Z","iopub.status.idle":"2022-07-12T12:49:37.160259Z","shell.execute_reply.started":"2022-07-12T12:49:36.689514Z","shell.execute_reply":"2022-07-12T12:49:37.159369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(rc={'figure.figsize':(11, 4)})\ntrain['humidity'].plot(linewidth=0.5);","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:49:40.888289Z","iopub.execute_input":"2022-07-12T12:49:40.888622Z","iopub.status.idle":"2022-07-12T12:49:41.155923Z","shell.execute_reply.started":"2022-07-12T12:49:40.888594Z","shell.execute_reply":"2022-07-12T12:49:41.154950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['atemp'].plot(linewidth=0.5);","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:49:44.913048Z","iopub.execute_input":"2022-07-12T12:49:44.913687Z","iopub.status.idle":"2022-07-12T12:49:45.216805Z","shell.execute_reply.started":"2022-07-12T12:49:44.913650Z","shell.execute_reply":"2022-07-12T12:49:45.215679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plott = ['temp', 'humidity', 'windspeed', 'atemp']\naxes = train[plott].plot(marker='.', alpha=0.5, linestyle='None', figsize=(11, 9), subplots=True)\nfor ax in axes:\n    ax.set_ylabel('Weather changes')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:49:49.316738Z","iopub.execute_input":"2022-07-12T12:49:49.317132Z","iopub.status.idle":"2022-07-12T12:49:50.089983Z","shell.execute_reply.started":"2022-07-12T12:49:49.317099Z","shell.execute_reply":"2022-07-12T12:49:50.089041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax = train.loc[train[\"year\"] == 2012, 'humidity'].plot()\nax.set_ylabel('Daily Humidity');","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:49:54.744932Z","iopub.execute_input":"2022-07-12T12:49:54.745493Z","iopub.status.idle":"2022-07-12T12:49:55.048922Z","shell.execute_reply.started":"2022-07-12T12:49:54.745459Z","shell.execute_reply":"2022-07-12T12:49:55.048056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax = train.loc[train[\"year\"] == 2012, 'humidity'].plot(marker='o', linestyle='-')\nax.set_ylabel('Daily Humidity in 2012');","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:50:14.260941Z","iopub.execute_input":"2022-07-12T12:50:14.261542Z","iopub.status.idle":"2022-07-12T12:50:14.495825Z","shell.execute_reply.started":"2022-07-12T12:50:14.261506Z","shell.execute_reply":"2022-07-12T12:50:14.494960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.set_index([\"datetime\"])\ntrain","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:50:18.527708Z","iopub.execute_input":"2022-07-12T12:50:18.528071Z","iopub.status.idle":"2022-07-12T12:50:18.561314Z","shell.execute_reply.started":"2022-07-12T12:50:18.528039Z","shell.execute_reply":"2022-07-12T12:50:18.560291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr = train.select_dtypes(\"number\").corr()\ncorr","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:50:29.314866Z","iopub.execute_input":"2022-07-12T12:50:29.315624Z","iopub.status.idle":"2022-07-12T12:50:29.393353Z","shell.execute_reply.started":"2022-07-12T12:50:29.315581Z","shell.execute_reply":"2022-07-12T12:50:29.391707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(corr);","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:48:05.971258Z","iopub.status.idle":"2022-07-12T12:48:05.972186Z","shell.execute_reply.started":"2022-07-12T12:48:05.971918Z","shell.execute_reply":"2022-07-12T12:48:05.971942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.pairplot(corr);","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:48:05.973386Z","iopub.status.idle":"2022-07-12T12:48:05.974196Z","shell.execute_reply.started":"2022-07-12T12:48:05.973937Z","shell.execute_reply":"2022-07-12T12:48:05.973961Z"},"trusted":true},"execution_count":null,"outputs":[]}]}