{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\n\n**The goal of this noteboos is to show you how to visualize Time Series(Tabular data) in different ways and also how to handle formatted date**.","metadata":{}},{"cell_type":"markdown","source":"## Objective\n​\n* Handling Date\n* Convert Information to Data\n* Univariate Analysis\n* Multivariate Analysis\n* Yearly Analysis\n* Monthly Analysis\n* Dayly Analysis\n\n**Using differnt graphs to Visualize the data**\n\n    * Different Bar plots\n    * Different line plots\n    * Histplot","metadata":{}},{"cell_type":"markdown","source":"# Set Up","metadata":{}},{"cell_type":"markdown","source":"<div style=\"color:#FFFFFF          ; display  :fill; border-radius:90px;\n           background-color:#13A4B4; font-size:10px; font-family  :cursive\">\n    \n<p style=\"padding   : .1px;    color    :#FFFFFF; \n          text-align: Left;   font-size:28px; font-family  :cursive\">       \n&emsp;&emsp;&emsp;&emsp;&emsp;&emsp;&emsp;&nbsp;Importing the Libraries<a id=5></a></p>\n</div>","metadata":{}},{"cell_type":"code","source":"# import classes for Data Analysis\nimport numpy  as np\nimport pandas as pd\n\n# import classes for data visulation\nimport matplotlib.pyplot    as plt\nimport seaborn              as sns\nimport plotly.express       as px\nimport plotly.graph_objects as go\nimport plotly.figure_factory as ff\n\nimport holoviews as hv\nfrom holoviews import opts\nhv.extension('bokeh')\n\nfrom fastai.tabular.all import *\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:09:42.132189Z","iopub.execute_input":"2022-06-06T09:09:42.132837Z","iopub.status.idle":"2022-06-06T09:09:44.715550Z","shell.execute_reply.started":"2022-06-06T09:09:42.132806Z","shell.execute_reply":"2022-06-06T09:09:44.714711Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:#FFFFFF          ; display  :fill; border-radius:90px;\n           background-color:#13A4B4; font-size:10px; font-family  :cursive\">\n    \n<p style=\"padding   : .1px;    color    :#FFFFFF; \n          text-align: Left;   font-size:28px; font-family  :cursive\">   \n&emsp;&emsp;&emsp;&emsp;&emsp;&emsp;&emsp;&nbsp; Importing the Dataset <a id=6></a></p>","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/weather-dataset/weatherHistory.csv')\n# view dimensions of dataset\nprint(\"view dimensions of dataset\")\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:09:44.716931Z","iopub.execute_input":"2022-06-06T09:09:44.717606Z","iopub.status.idle":"2022-06-06T09:09:45.209046Z","shell.execute_reply.started":"2022-06-06T09:09:44.717575Z","shell.execute_reply":"2022-06-06T09:09:45.208199Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preview the dataset\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:09:47.339183Z","iopub.execute_input":"2022-06-06T09:09:47.340108Z","iopub.status.idle":"2022-06-06T09:09:47.368005Z","shell.execute_reply.started":"2022-06-06T09:09:47.340052Z","shell.execute_reply":"2022-06-06T09:09:47.367040Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preparation","metadata":{}},{"cell_type":"markdown","source":"<div style=\"color:#FFFFFF          ; display  :fill; border-radius:90px;\n           background-color:#13A4B4; font-size:10px; font-family  :cursive\">\n    \n<p style=\"padding   : .1px;    color    :#FFFFFF; \n          text-align: Left;   font-size:28px; font-family  :cursive\">          \n&emsp;&emsp;&emsp;&emsp;&emsp;&emsp;&nbsp;&nbsp; Missing Value Detection  <a id=10></a>\n</p>\n</div>","metadata":{}},{"cell_type":"code","source":"plt.style.use('seaborn')\nplt.figure(figsize = (10,5))\nsns.heatmap(df.isnull(), yticklabels = False, cmap = 'plasma')\nplt.title('Null Values in Data Frame')","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:09:59.016261Z","iopub.execute_input":"2022-06-06T09:09:59.016625Z","iopub.status.idle":"2022-06-06T09:10:00.436679Z","shell.execute_reply.started":"2022-06-06T09:09:59.016598Z","shell.execute_reply":"2022-06-06T09:10:00.433713Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get the number of missing data points per column\nmissing_value_count = (df.isnull().sum())\nprint(missing_value_count[missing_value_count > 0])\n# percentage of missing data\ntotal_cells = np.product(df.shape)\ntotal_missing_value = missing_value_count.sum()\nprint(\"Total percentage of our missing value is:\",round((total_missing_value / total_cells * 100),4))\nprint('Total number of our cells is :',total_cells)\nprint('Total number of our missing value is :',total_missing_value)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:10:00.438143Z","iopub.execute_input":"2022-06-06T09:10:00.438642Z","iopub.status.idle":"2022-06-06T09:10:00.491816Z","shell.execute_reply.started":"2022-06-06T09:10:00.438610Z","shell.execute_reply":"2022-06-06T09:10:00.490864Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:#FFFFFF          ; display  :fill; border-radius:90px;\n           background-color:#13A4B4; font-size:10px; font-family  :cursive\">\n    \n<p style=\"padding   : .1px;    color    :#FFFFFF; \n          text-align: Left;   font-size:28px; font-family  :cursive\">          \n&emsp;&emsp;&emsp;&emsp;&emsp;&emsp;&nbsp;&nbsp;Handling missing values  <a id=10></a>\n</p>\n</div>","metadata":{}},{"cell_type":"code","source":"df['Precip Type'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:10:03.800942Z","iopub.execute_input":"2022-06-06T09:10:03.801511Z","iopub.status.idle":"2022-06-06T09:10:03.827936Z","shell.execute_reply.started":"2022-06-06T09:10:03.801477Z","shell.execute_reply":"2022-06-06T09:10:03.827041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* As we can see that we have 517 missing values and unfortunately we cann't **delete it**, because removing 517 index means removing 517 **random Formatted Date** which is canna have **large negative impact** on our Analysis.\n* As we can see that the majorty of Precip Type is rain so that why we will replace the missing values with rain variable.","metadata":{}},{"cell_type":"code","source":"# replace the missing values with rain variable\ndf['Precip Type'].fillna(\"rain\", inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:10:04.983543Z","iopub.execute_input":"2022-06-06T09:10:04.983933Z","iopub.status.idle":"2022-06-06T09:10:05.000626Z","shell.execute_reply.started":"2022-06-06T09:10:04.983904Z","shell.execute_reply":"2022-06-06T09:10:04.999702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* So after filling the missing values, let's check it more again:","metadata":{}},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:10:06.596777Z","iopub.execute_input":"2022-06-06T09:10:06.597378Z","iopub.status.idle":"2022-06-06T09:10:06.649941Z","shell.execute_reply.started":"2022-06-06T09:10:06.597334Z","shell.execute_reply":"2022-06-06T09:10:06.648941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* So Now let's delete 'Daily Summary', we do not need it because we do have 'Summary' feature","metadata":{}},{"cell_type":"code","source":"df.drop([\"Daily Summary\"], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:10:07.724855Z","iopub.execute_input":"2022-06-06T09:10:07.725272Z","iopub.status.idle":"2022-06-06T09:10:07.737525Z","shell.execute_reply.started":"2022-06-06T09:10:07.725239Z","shell.execute_reply":"2022-06-06T09:10:07.736645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Let's check it:","metadata":{}},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:10:18.425692Z","iopub.execute_input":"2022-06-06T09:10:18.426312Z","iopub.status.idle":"2022-06-06T09:10:18.433586Z","shell.execute_reply.started":"2022-06-06T09:10:18.426264Z","shell.execute_reply":"2022-06-06T09:10:18.432524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"<div style=\"color:#FFFFFF          ; display  :fill; border-radius:90px;\n           background-color:#13A4B4; font-size:10px; font-family  :cursive\">\n    \n<p style=\"padding   : .1px;    color    :#FFFFFF; \n          text-align: Left;   font-size:28px; font-family  :cursive\">          \n&emsp;&emsp;&emsp;&emsp;&emsp;&emsp;&nbsp;&nbsp;Handling Dates  <a id=10></a>\n</p>\n</div>","metadata":{}},{"cell_type":"code","source":"#Changing Formatted Date from String to Datetime\ndf['Formatted Date'] = pd.to_datetime(df['Formatted Date'],utc=True)\ndf['Formatted Date'][0]","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:10:25.098520Z","iopub.execute_input":"2022-06-06T09:10:25.098913Z","iopub.status.idle":"2022-06-06T09:10:25.913002Z","shell.execute_reply.started":"2022-06-06T09:10:25.098883Z","shell.execute_reply":"2022-06-06T09:10:25.912227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.sort_values(by='Formatted Date')","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:10:51.036935Z","iopub.execute_input":"2022-06-06T09:10:51.037347Z","iopub.status.idle":"2022-06-06T09:10:51.054384Z","shell.execute_reply.started":"2022-06-06T09:10:51.037315Z","shell.execute_reply":"2022-06-06T09:10:51.053678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:#FFFFFF          ; display  :fill; border-radius:90px;\n           background-color:#13A4B4; font-size:10px; font-family  :cursive\">\n    \n<p style=\"padding   : .1px;    color    :#FFFFFF; \n          text-align: Left;   font-size:28px; font-family  :cursive\">          \n&emsp;&emsp;&emsp;&emsp;&emsp;&emsp;&nbsp;&nbsp;Create monthly and daily data frame  <a id=10></a>\n</p>\n</div>","metadata":{}},{"cell_type":"code","source":"# this DataFrame will use Formatted Date as an index just for visualization objective\ndata = df.copy()\n#data['Formatted Date'] = data['Formatted Date'].strftime(\"%d/%m/%Y\")\n#print('Date String:', data['Formatted Date'][0])\n# Set 'Formatted Date' as index and resample Date\nresampled_df = data.set_index('Formatted Date')\nresampled_days = resampled_df.resample('D').mean()\nresampled_df = resampled_df.resample('M').mean()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:13:35.601931Z","iopub.execute_input":"2022-06-06T09:13:35.603476Z","iopub.status.idle":"2022-06-06T09:13:35.661256Z","shell.execute_reply.started":"2022-06-06T09:13:35.603405Z","shell.execute_reply":"2022-06-06T09:13:35.660208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(resampled_df),len(df),len(resampled_days)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:13:36.223351Z","iopub.execute_input":"2022-06-06T09:13:36.224154Z","iopub.status.idle":"2022-06-06T09:13:36.230832Z","shell.execute_reply.started":"2022-06-06T09:13:36.224117Z","shell.execute_reply":"2022-06-06T09:13:36.229971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"resampled_df.head(1)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:13:41.799480Z","iopub.execute_input":"2022-06-06T09:13:41.800245Z","iopub.status.idle":"2022-06-06T09:13:41.825232Z","shell.execute_reply.started":"2022-06-06T09:13:41.800196Z","shell.execute_reply":"2022-06-06T09:13:41.824074Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Resample('M')** simply converting the hourly data to monthly by taking the mean. But it will also drop all the \"object\" columns. SO what we canna do later is conacting these \"object\" columns with all kind of Formatted Date cloumns in one Data Frame (for a visualization purpose).","metadata":{}},{"cell_type":"markdown","source":"And **Resample('D')** does the exact same idea but it convert the time to daily data","metadata":{}},{"cell_type":"markdown","source":"Now after changing Formatted Date from String to Datetime we will replace every date column with a set of date metadata columns, such as holiday, day of week, and month. These columns provide categorical data that we suspect will be useful.\n\nfastai comes with a function that will do this for us—we just have to pass a column name that contains dates:","metadata":{}},{"cell_type":"code","source":"from fastai.tabular.all import *\n# make a Date copy because \"add_datepart\" do delete the orginal formatted Date.\n#In other word the function transfer the column to Date Parts\ndf['Date'] = df[\"Formatted Date\"]\ndf = add_datepart(df, 'Formatted Date')","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:16:06.053084Z","iopub.execute_input":"2022-06-06T09:16:06.053591Z","iopub.status.idle":"2022-06-06T09:16:06.241376Z","shell.execute_reply.started":"2022-06-06T09:16:06.053553Z","shell.execute_reply":"2022-06-06T09:16:06.240386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[2000:2001]","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:16:22.525081Z","iopub.execute_input":"2022-06-06T09:16:22.525619Z","iopub.status.idle":"2022-06-06T09:16:22.552656Z","shell.execute_reply.started":"2022-06-06T09:16:22.525574Z","shell.execute_reply":"2022-06-06T09:16:22.551503Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that there are now lots of new columns in our DataFrame:","metadata":{}},{"cell_type":"code","source":"' '.join(o for o in df.columns if o.startswith('Formatted'))","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:16:34.990903Z","iopub.execute_input":"2022-06-06T09:16:34.991418Z","iopub.status.idle":"2022-06-06T09:16:34.998574Z","shell.execute_reply.started":"2022-06-06T09:16:34.991382Z","shell.execute_reply":"2022-06-06T09:16:34.997725Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:16:35.529368Z","iopub.execute_input":"2022-06-06T09:16:35.530487Z","iopub.status.idle":"2022-06-06T09:16:35.572694Z","shell.execute_reply.started":"2022-06-06T09:16:35.530438Z","shell.execute_reply":"2022-06-06T09:16:35.571752Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:#FFFFFF          ; display  :fill; border-radius:90px;\n           background-color:#13A4B4; font-size:10px; font-family  :cursive\">\n    \n<p style=\"padding   : .1px;    color    :#FFFFFF; \n          text-align: Left;   font-size:28px; font-family  :cursive\">          \n&emsp;&emsp;&emsp;&emsp;&emsp;&emsp;&nbsp;&nbsp;Convert 'bool' to 'object'  <a id=10></a>\n</p>\n</div>","metadata":{}},{"cell_type":"code","source":"df[['Formatted Is_month_end','Formatted Is_month_start','Formatted Is_quarter_end',\n            'Formatted Is_quarter_start','Formatted Is_year_end','Formatted Is_year_start']].head(1)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:16:38.366160Z","iopub.execute_input":"2022-06-06T09:16:38.367291Z","iopub.status.idle":"2022-06-06T09:16:38.380415Z","shell.execute_reply.started":"2022-06-06T09:16:38.367210Z","shell.execute_reply":"2022-06-06T09:16:38.379450Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bool_list = ['Formatted Is_month_end','Formatted Is_month_start','Formatted Is_quarter_end',\n            'Formatted Is_quarter_start','Formatted Is_year_end','Formatted Is_year_start']","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:16:38.939566Z","iopub.execute_input":"2022-06-06T09:16:38.940335Z","iopub.status.idle":"2022-06-06T09:16:38.945782Z","shell.execute_reply.started":"2022-06-06T09:16:38.940289Z","shell.execute_reply":"2022-06-06T09:16:38.944552Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for var in bool_list:\n    df[var] = (df[var].map({True:'Yes', False:'No'}))\n    \ndf.info()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:16:39.784276Z","iopub.execute_input":"2022-06-06T09:16:39.784658Z","iopub.status.idle":"2022-06-06T09:16:39.971764Z","shell.execute_reply.started":"2022-06-06T09:16:39.784627Z","shell.execute_reply":"2022-06-06T09:16:39.970550Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[['Formatted Is_month_end','Formatted Is_month_start','Formatted Is_quarter_end',\n            'Formatted Is_quarter_start','Formatted Is_year_end','Formatted Is_year_start']].head(1)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:16:44.986645Z","iopub.execute_input":"2022-06-06T09:16:44.987019Z","iopub.status.idle":"2022-06-06T09:16:45.012493Z","shell.execute_reply.started":"2022-06-06T09:16:44.986988Z","shell.execute_reply":"2022-06-06T09:16:45.011799Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# view the new dimensions of dataset\nprint(\"The new shape of dataset is:\",df.shape)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:16:46.282179Z","iopub.execute_input":"2022-06-06T09:16:46.282892Z","iopub.status.idle":"2022-06-06T09:16:46.287499Z","shell.execute_reply.started":"2022-06-06T09:16:46.282855Z","shell.execute_reply":"2022-06-06T09:16:46.286848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Converting Information to Data","metadata":{}},{"cell_type":"markdown","source":"## Seasonal information\n><div class=\"alert alert-success\" role=\"alert\">\n>Let's assume this data was collected in Europe.<br/>\n>select the four climatological seasons as below. I think the most eurpe countries have these seasons time.\n><ul>\n>    <li><b>Winter</b> : December to March</li>\n>    <li><b>Summer</b> : June to August</li>\n>    <li><b>Spring</b> : April to May</li>\n>    <li><b>Autumn</b> : September to November</li>\n></ul>\n>We can create seasonal variable based on month variable.<br/>\n\n></div>","metadata":{}},{"cell_type":"code","source":"def month2seasons(x):\n    if x in [12, 1, 2, 3]:\n        season = 'Winter'\n    elif x in [6, 7, 8]:\n        season = 'Summer'\n    elif x in [4, 5]:\n        season = 'Spring '\n    elif x in [9, 10, 11]:\n        season = 'Autumn'\n    return season","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:44:43.308929Z","iopub.execute_input":"2022-06-06T09:44:43.309959Z","iopub.status.idle":"2022-06-06T09:44:43.317322Z","shell.execute_reply.started":"2022-06-06T09:44:43.309913Z","shell.execute_reply":"2022-06-06T09:44:43.316468Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['season'] = df['Formatted Month'].apply(month2seasons)\ndf['season'].head(3)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:44:43.715326Z","iopub.execute_input":"2022-06-06T09:44:43.715859Z","iopub.status.idle":"2022-06-06T09:44:43.787259Z","shell.execute_reply.started":"2022-06-06T09:44:43.715814Z","shell.execute_reply":"2022-06-06T09:44:43.786324Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Timing information\n><div class=\"alert alert-success\" role=\"alert\">\n>Hour variable can be broken into Night, Morning, Afternoon and Evening based on its number.\n><ul>\n>    <li><b>Night</b> : 22:00 - 23:59 / 00:00 - 03:59</li>\n>    <li><b>Morning</b> : 04:00 - 11:59</li>\n>    <li><b>Afternoon</b> : 12:00 - 16:59</li>\n>    <li><b>Evening</b> : 17:00 - 21:59</li>\n></ul>\n>We can create timing variable based on hour variable.<br/>\n</u>\n></div>","metadata":{}},{"cell_type":"code","source":"def hours2timing(x):\n    if x in [22,23,0,1,2,3]:\n        timing = 'Night'\n    elif x in range(4, 12):\n        timing = 'Morning'\n    elif x in range(12, 17):\n        timing = 'Afternoon'\n    elif x in range(17, 22):\n        timing = 'Evening'\n    else:\n        timing = 'X'\n    return timing","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:44:45.093472Z","iopub.execute_input":"2022-06-06T09:44:45.093858Z","iopub.status.idle":"2022-06-06T09:44:45.100488Z","shell.execute_reply.started":"2022-06-06T09:44:45.093825Z","shell.execute_reply":"2022-06-06T09:44:45.099486Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['hour'] = df['Date'].apply(lambda x : x.hour)\ndf['timing'] = df['hour'].apply(hours2timing)\ndf['timing'].head(3)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:44:53.368310Z","iopub.execute_input":"2022-06-06T09:44:53.368676Z","iopub.status.idle":"2022-06-06T09:44:53.485212Z","shell.execute_reply.started":"2022-06-06T09:44:53.368648Z","shell.execute_reply":"2022-06-06T09:44:53.484298Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# view the new dimensions of dataset\nprint(\"The new shape of dataset is:\",df.shape)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:44:53.974239Z","iopub.execute_input":"2022-06-06T09:44:53.974740Z","iopub.status.idle":"2022-06-06T09:44:53.980531Z","shell.execute_reply.started":"2022-06-06T09:44:53.974687Z","shell.execute_reply":"2022-06-06T09:44:53.979494Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Visualization","metadata":{}},{"cell_type":"markdown","source":"<div style=\"color:#FFFFFF          ; display  :fill; border-radius:90px;\n           background-color:#13A4B4; font-size:10px; font-family  :cursive\">\n    \n<p style=\"padding   : .1px;    color    :#FFFFFF; \n          text-align: Left;   font-size:28px; font-family  :cursive\">          \n&emsp;&emsp;&emsp;&emsp;&emsp;&emsp;&nbsp;&nbsp; Univariate Analysis <a id=10></a>\n</p>\n</div>","metadata":{}},{"cell_type":"markdown","source":"### Temperature\n>Temperature clearly consists of multiple distributions.","metadata":{}},{"cell_type":"code","source":"hv.Distribution(df['Temperature (C)']).opts(title=\"Temperature Distribution\", color=\"green\", xlabel=\"Temperature\", ylabel=\"Density\")\\\n                            .opts(opts.Distribution(width=700, height=300,tools=['hover'],show_grid=True))","metadata":{"execution":{"iopub.status.busy":"2022-06-06T10:17:55.571696Z","iopub.execute_input":"2022-06-06T10:17:55.572152Z","iopub.status.idle":"2022-06-06T10:17:55.997519Z","shell.execute_reply.started":"2022-06-06T10:17:55.572115Z","shell.execute_reply":"2022-06-06T10:17:55.996045Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I recommend to use hist plots for each numerical feature in the data frame to see how our data are distributed. it's very helpful with extracting information","metadata":{}},{"cell_type":"markdown","source":"#### Season","metadata":{}},{"cell_type":"code","source":"season_cnt = np.round(df['season'].value_counts(normalize=True) * 100)\nhv.Bars(season_cnt).opts(title=\"Season Count\", color=\"red\", xlabel=\"Season\", ylabel=\"Percentage\", yformatter='%d%%')\\\n                .opts(opts.Bars(width=700, height=300,tools=['hover'],show_grid=True))","metadata":{"execution":{"iopub.status.busy":"2022-06-06T10:22:17.838117Z","iopub.execute_input":"2022-06-06T10:22:17.838661Z","iopub.status.idle":"2022-06-06T10:22:18.001797Z","shell.execute_reply.started":"2022-06-06T10:22:17.838616Z","shell.execute_reply":"2022-06-06T10:22:18.001139Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Timing","metadata":{}},{"cell_type":"code","source":"timing_cnt = np.round(df['timing'].value_counts(normalize=True) * 100)\nhv.Bars(timing_cnt).opts(title=\"Timing Count\", color=\"black\", xlabel=\"Timing\", ylabel=\"Percentage\", yformatter='%d%%')\\\n                .opts(opts.Bars(width=700, height=300,tools=['hover'],show_grid=True))","metadata":{"execution":{"iopub.status.busy":"2022-06-06T10:26:27.435644Z","iopub.execute_input":"2022-06-06T10:26:27.436093Z","iopub.status.idle":"2022-06-06T10:26:27.590772Z","shell.execute_reply.started":"2022-06-06T10:26:27.436051Z","shell.execute_reply":"2022-06-06T10:26:27.590160Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:#FFFFFF          ; display  :fill; border-radius:90px;\n           background-color:#13A4B4; font-size:10px; font-family  :cursive\">\n    \n<p style=\"padding   : .1px;    color    :#FFFFFF; \n          text-align: Left;   font-size:28px; font-family  :cursive\">          \n&emsp;&emsp;&emsp;&emsp;&emsp;&emsp;&nbsp;&nbsp; Multivariate Analysis <a id=10></a>\n</p>\n</div>","metadata":{}},{"cell_type":"markdown","source":"#### Temperature by Season","metadata":{}},{"cell_type":"code","source":"season_agg = df.groupby('season').agg({'Temperature (C)': ['min', 'max']})\nseason_maxmin = pd.merge(season_agg['Temperature (C)']['max'],season_agg['Temperature (C)']['min'],right_index=True,left_index=True)\nseason_maxmin = pd.melt(season_maxmin.reset_index(), ['season']).rename(columns={'season':'Season', 'variable':'Max/Min'})\nhv.Bars(season_maxmin, ['Season', 'Max/Min'], 'value').opts(title=\"Temperature by Season Max/Min\", ylabel=\"Temperature\")\\\n                                                                    .opts(opts.Bars(width=700, height=300,tools=['hover'],show_grid=True))","metadata":{"execution":{"iopub.status.busy":"2022-06-06T10:28:02.541775Z","iopub.execute_input":"2022-06-06T10:28:02.542329Z","iopub.status.idle":"2022-06-06T10:28:02.734614Z","shell.execute_reply.started":"2022-06-06T10:28:02.542284Z","shell.execute_reply":"2022-06-06T10:28:02.733543Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Temperature by Timing","metadata":{}},{"cell_type":"code","source":"timing_agg = df.groupby('timing').agg({'Temperature (C)': ['min', 'max']})\ntiming_maxmin = pd.merge(timing_agg['Temperature (C)']['max'],timing_agg['Temperature (C)']['min'],right_index=True,left_index=True)\ntiming_maxmin = pd.melt(timing_maxmin.reset_index(), ['timing']).rename(columns={'timing':'Timing', 'variable':'Max/Min'})\nhv.Bars(timing_maxmin, ['Timing', 'Max/Min'], 'value').opts(title=\"Temperature by Timing Max/Min\", ylabel=\"Temperature\")\\\n                                                                    .opts(opts.Bars(width=700, height=300,tools=['hover'],show_grid=True))","metadata":{"execution":{"iopub.status.busy":"2022-06-06T10:29:07.875915Z","iopub.execute_input":"2022-06-06T10:29:07.876313Z","iopub.status.idle":"2022-06-06T10:29:08.061746Z","shell.execute_reply.started":"2022-06-06T10:29:07.876283Z","shell.execute_reply":"2022-06-06T10:29:08.060832Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.line(resampled_df,x=resampled_df.index,y=['Temperature (C)','Humidity','Wind Speed (km/h)','Visibility (km)','Loud Cover','Pressure (millibars)'],title='All features distribution')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T10:37:40.562223Z","iopub.execute_input":"2022-06-06T10:37:40.562643Z","iopub.status.idle":"2022-06-06T10:37:40.690886Z","shell.execute_reply.started":"2022-06-06T10:37:40.562612Z","shell.execute_reply":"2022-06-06T10:37:40.689910Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.line(resampled_df,x=resampled_df.index,y=['Temperature (C)','Humidity','Wind Speed (km/h)','Visibility (km)','Loud Cover'],title='All features distribution withot Pressure (millibars)')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T10:38:46.021225Z","iopub.execute_input":"2022-06-06T10:38:46.022328Z","iopub.status.idle":"2022-06-06T10:38:46.140619Z","shell.execute_reply.started":"2022-06-06T10:38:46.022277Z","shell.execute_reply":"2022-06-06T10:38:46.139727Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"See the different, always try to combine the features that does not have much difference in its values,In other words, make sure that they are distributed in a limited range sufficient to display all the features in a clear picture","metadata":{}},{"cell_type":"markdown","source":"**Now if you want to extract more information from the above chart or highlight specific features, I recommend you to use dist charts, it is one of the best ways to see the distribution of different features.**\n\nso let's see:","metadata":{}},{"cell_type":"code","source":"hist_data = [resampled_df['Temperature (C)'].values, resampled_df['Wind Speed (km/h)'].values, resampled_df['Visibility (km)'].values]\n\ngroup_labels = ['Temperature (C)', 'Wind Speed (km/h)', 'Visibility (km)']\ncolors = ['#A56CC1', '#A6ACEC', '#63F5EF']\n\n# Create distplot with curve_type set to 'normal'\nfig = ff.create_distplot(hist_data, group_labels, colors=colors,\n                         bin_size=.2, show_rug=False)\n\n# Add title\nfig.update_layout(title_text='Hist and Curve Plot')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T10:58:48.794151Z","iopub.execute_input":"2022-06-06T10:58:48.794564Z","iopub.status.idle":"2022-06-06T10:58:48.833332Z","shell.execute_reply.started":"2022-06-06T10:58:48.794529Z","shell.execute_reply":"2022-06-06T10:58:48.832402Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"in_month = df[df['Formatted Is_month_start']=='Yes'].groupby('Formatted Month').agg({'Humidity':['mean']})\nin_month.columns = [f\"{i[0]}_{i[1]}\" for i in in_month.columns]\nout_month = df[df['Formatted Is_month_start']=='No'].groupby('Formatted Month').agg({'Humidity':['mean']})\nout_month.columns = [f\"{i[0]}_{i[1]}\" for i in out_month.columns]\nhv.Curve(in_month, label='Yes') * hv.Curve(out_month, label='No').opts(\n          title=\"Average Humidity by month start\", ylabel=\"Humidity\", xlabel='Month')\\\n         .opts(opts.Curve(width=700, height=300,tools=['hover'],show_grid=True))","metadata":{"execution":{"iopub.status.busy":"2022-06-06T10:54:56.271372Z","iopub.execute_input":"2022-06-06T10:54:56.271776Z","iopub.status.idle":"2022-06-06T10:54:56.577275Z","shell.execute_reply.started":"2022-06-06T10:54:56.271742Z","shell.execute_reply":"2022-06-06T10:54:56.576243Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:#FFFFFF          ; display  :fill; border-radius:90px;\n           background-color:#13A4B4; font-size:10px; font-family  :cursive\">\n    \n<p style=\"padding   : .1px;    color    :#FFFFFF; \n          text-align: Left;   font-size:28px; font-family  :cursive\"> \n&emsp; Line Vs Bar (plots) <a id=13></a>  \n</p>\n    \n<p> </p>\n    \n</div>\n\n<div style=\"color:#000000          ; display  :fill; border-radius:90px;\n           background-color:#FF8000; font-size:20px; font-family  :cursive\">\n    \n<p style=\"padding   : .1px;    color    :#FFFFFF; \n          text-align: Left;   font-size:24px; font-family  :cursive\">          \n&emsp;&nbsp;&nbsp;&emsp;&emsp;&emsp;&emsp;&emsp;&nbsp;Monthly  <a id=14></a>  \n</p>\n</div>","metadata":{}},{"cell_type":"code","source":"fig = px.line(resampled_df, x=resampled_df.index, y=\"Temperature (C)\", title = \"Average Monthly Temperature (C)\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:49:06.883770Z","iopub.execute_input":"2022-06-06T09:49:06.884220Z","iopub.status.idle":"2022-06-06T09:49:06.991605Z","shell.execute_reply.started":"2022-06-06T09:49:06.884185Z","shell.execute_reply":"2022-06-06T09:49:06.990720Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.bar(resampled_df, x=resampled_df.index, y=\"Temperature (C)\", title = \"Average Monthly Temperature (C)\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:50:40.378163Z","iopub.execute_input":"2022-06-06T09:50:40.378749Z","iopub.status.idle":"2022-06-06T09:50:40.475096Z","shell.execute_reply.started":"2022-06-06T09:50:40.378703Z","shell.execute_reply":"2022-06-06T09:50:40.474043Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can see from the graphs above that bars make negative values stand out, that mean in this situation is going to be better to use bar plot","metadata":{}},{"cell_type":"code","source":"fig = px.bar(resampled_df, x=resampled_df.index, y='Humidity', title = \"Average Monthly Humidity\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:59:28.894700Z","iopub.execute_input":"2022-06-06T09:59:28.895155Z","iopub.status.idle":"2022-06-06T09:59:28.973722Z","shell.execute_reply.started":"2022-06-06T09:59:28.895121Z","shell.execute_reply":"2022-06-06T09:59:28.972814Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.line(resampled_df, x=resampled_df.index, y='Humidity',  title = \"Average Monthly Humidity\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T09:59:30.880573Z","iopub.execute_input":"2022-06-06T09:59:30.881140Z","iopub.status.idle":"2022-06-06T09:59:30.953765Z","shell.execute_reply.started":"2022-06-06T09:59:30.881106Z","shell.execute_reply":"2022-06-06T09:59:30.952706Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" As we can see in bar plot because there is no negative values in Humidity the graph doesn't display the values very well. \n If you have a seen the line plot above it does display Humidity very well and  you can understand and extract much more information better than the bar plot.\n \n in this situation, i recommend you to use **min()** and **max()** for the feature that you are about display before starting ploting.","metadata":{}},{"cell_type":"markdown","source":"<div style=\"color:#FFFFFF          ; display  :fill; border-radius:90px;\n           background-color:#13A4B4; font-size:10px; font-family  :cursive\">\n    \n<p style=\"padding   : .1px;    color    :#FFFFFF; \n          text-align: Left;   font-size:28px; font-family  :cursive\">          \n&emsp;&emsp;&emsp;&emsp;&emsp;&emsp;&nbsp;&nbsp; Dayly Visualization <a id=10></a>\n</p>\n</div>","metadata":{}},{"cell_type":"code","source":"# just make it shorter to write\ndayly_df = resampled_days\nfig = px.line(dayly_df, x = dayly_df.index, y = \"Pressure (millibars)\",title = \"Average Dayly Pressure (millibars) over the year\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T11:13:08.344643Z","iopub.execute_input":"2022-06-06T11:13:08.345471Z","iopub.status.idle":"2022-06-06T11:13:08.567476Z","shell.execute_reply.started":"2022-06-06T11:13:08.345421Z","shell.execute_reply":"2022-06-06T11:13:08.566821Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.line(dayly_df, x = dayly_df.index, y = \"Visibility (km)\",title = \"Average Dayly Visibility (km)\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T11:14:30.038141Z","iopub.execute_input":"2022-06-06T11:14:30.038505Z","iopub.status.idle":"2022-06-06T11:14:30.247915Z","shell.execute_reply.started":"2022-06-06T11:14:30.038477Z","shell.execute_reply":"2022-06-06T11:14:30.247190Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Stylish line plot","metadata":{}},{"cell_type":"code","source":"fig = px.line(resampled_df, x=resampled_df.index, y='Humidity', title='Time Series with Range Slider and Selectors')\n\nfig.update_xaxes(\n    rangeslider_visible=True,\n    rangeselector=dict(\n        buttons=list([\n            dict(count=1, label=\"1m\", step=\"month\", stepmode=\"backward\"),\n            dict(count=6, label=\"6m\", step=\"month\", stepmode=\"backward\"),\n            dict(count=1, label=\"YTD\", step=\"year\", stepmode=\"todate\"),\n            dict(count=1, label=\"1y\", step=\"year\", stepmode=\"backward\"),\n            dict(step=\"all\")\n        ])\n    )\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T11:15:49.086583Z","iopub.execute_input":"2022-06-06T11:15:49.087055Z","iopub.status.idle":"2022-06-06T11:15:49.183711Z","shell.execute_reply.started":"2022-06-06T11:15:49.087001Z","shell.execute_reply":"2022-06-06T11:15:49.182815Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def temperature_graph(df):\n    df = df.sort_values('Date')\n    fig = go.Figure()\n    fig.add_trace(go.Scatter(x=df['Date'], y=df['Temperature (C)'],  line_color='lightskyblue', opacity=0.7))\n    fig.update_layout(template='plotly_dark',title_text='Temperature (C)', xaxis_rangeslider_visible=True)\n    fig.show()\n    \ntemperature_graph(df)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T11:17:35.416812Z","iopub.execute_input":"2022-06-06T11:17:35.417175Z","iopub.status.idle":"2022-06-06T11:17:38.540658Z","shell.execute_reply.started":"2022-06-06T11:17:35.417144Z","shell.execute_reply":"2022-06-06T11:17:38.539144Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create figure and plot space\nfig, ax = plt.subplots(figsize=(10, 10))\n\n# Add x-axis and y-axis\nax.plot(resampled_df.index,\n        resampled_df['Temperature (C)'],\n        color='purple')\n\n# Set title and labels for axes\nax.set(xlabel=\"Date\",\n       ylabel='Temperature (C)',\n       title=\"Monthly Temperature (C) over the years\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T10:13:02.357734Z","iopub.execute_input":"2022-06-06T10:13:02.358180Z","iopub.status.idle":"2022-06-06T10:13:02.623968Z","shell.execute_reply.started":"2022-06-06T10:13:02.358145Z","shell.execute_reply":"2022-06-06T10:13:02.623309Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:#FFFFFF          ; display  :fill; border-radius:90px;\n           background-color:#13A4B4; font-size:10px; font-family  :cursive\">\n    \n<p style=\"padding   : .1px;    color    :#FFFFFF; \n          text-align: Left;   font-size:28px; font-family  :cursive\">          \n&emsp;&emsp;&emsp;&emsp;&emsp;&emsp;&nbsp;&nbsp; I hope you liked it <a id=10></a>\n</p>\n</div>","metadata":{}}]}