{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Libraries\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\nimport plotly.express as px","metadata":{"execution":{"iopub.status.busy":"2022-08-04T10:09:59.612007Z","iopub.execute_input":"2022-08-04T10:09:59.612459Z","iopub.status.idle":"2022-08-04T10:09:59.621315Z","shell.execute_reply.started":"2022-08-04T10:09:59.612428Z","shell.execute_reply":"2022-08-04T10:09:59.619527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#importing data\ndf_train = pd.read_csv('../input/Covid19-Death-Predictions/train.csv')\ndf_test = pd.read_csv('../input/Covid19-Death-Predictions/test.csv')\ndf = pd.concat([df_train,df_test]).reset_index()\ndisplay(df.head(), df.info())","metadata":{"execution":{"iopub.status.busy":"2022-08-04T09:53:38.725946Z","iopub.execute_input":"2022-08-04T09:53:38.726290Z","iopub.status.idle":"2022-08-04T09:53:39.181946Z","shell.execute_reply.started":"2022-08-04T09:53:38.726263Z","shell.execute_reply":"2022-08-04T09:53:39.180492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#removing unwanted columns\ndf = df.drop(['index','Id'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T09:53:39.184547Z","iopub.execute_input":"2022-08-04T09:53:39.185737Z","iopub.status.idle":"2022-08-04T09:53:39.208008Z","shell.execute_reply.started":"2022-08-04T09:53:39.185694Z","shell.execute_reply":"2022-08-04T09:53:39.206615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# there are lot of missing values in the data,to make data more accurate we have to first divide \n# data on basis of year then on basis of location then we will fill NAN values\n\n# now we are dividing data into three cateogries on basis of years\ndf_20 = df[df['Year']==2020]\ndf_20 = df_20.drop(['Year'],axis=1)\n\ndf_21 = df[df['Year']==2021]\ndf_21 = df_21.drop(['Year'],axis=1)\n\ndf_22 = df[df['Year']==2022]\ndf_22 = df_22.drop(['Year'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T09:53:39.210015Z","iopub.execute_input":"2022-08-04T09:53:39.210445Z","iopub.status.idle":"2022-08-04T09:53:39.251236Z","shell.execute_reply.started":"2022-08-04T09:53:39.210404Z","shell.execute_reply":"2022-08-04T09:53:39.249846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# now first we fill visualize the nu,ber of missing values in all the three years\nx1 = list(df_20.columns)\nfig = go.Figure()\nfig.add_trace(go.Bar(x=x1,\n                y=list(df_20.isna().sum()),\n                name='2020',\n                marker_color='rgb(55, 83, 109)'\n                ))\nfig.add_trace(go.Bar(x=x1,\n               y=list(df_21.isna().sum()),\n                name='2021',\n                marker_color='rgb(26, 118, 255)'\n                ))\nfig.add_trace(go.Bar(x=x1,\n                y=list(df_22.isna().sum()),\n                name='2022 ',\n                marker_color='rgb(30, 100, 125)'\n                ))\n\nfig.update_layout(\n    title='Missing values in features',\n    xaxis_tickfont_size=14,\n    yaxis=dict(\n        title='No of values missing ',\n        titlefont_size=16,\n        tickfont_size=14,\n    ),\n    legend=dict(\n        x=0,\n        y=1.0,\n        bgcolor='rgba(255, 255, 255, 0)',\n        bordercolor='rgba(255, 255, 255, 0)'\n    ),\n    barmode='group',\n    bargap=0.2,\n    bargroupgap=0.1\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T09:53:39.255195Z","iopub.execute_input":"2022-08-04T09:53:39.255715Z","iopub.status.idle":"2022-08-04T09:53:39.307292Z","shell.execute_reply.started":"2022-08-04T09:53:39.255673Z","shell.execute_reply":"2022-08-04T09:53:39.305858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from above barchart we can clearly see that the all three years holds different perspective of data\n# like 2020 data can only help us in getting insights of features like 'weekly cases','weekly deaths'and next week deaths\n# because rest of features in 2020 have more then 80 percent of missing values\n\n# and similarly the year 2021 and 2022 have different insights","metadata":{"execution":{"iopub.status.busy":"2022-08-04T09:53:39.309061Z","iopub.execute_input":"2022-08-04T09:53:39.310138Z","iopub.status.idle":"2022-08-04T09:53:39.317235Z","shell.execute_reply.started":"2022-08-04T09:53:39.310093Z","shell.execute_reply":"2022-08-04T09:53:39.315593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# now we will visualize the year first 2020\n# with help of pie chart we will see that how much percent of missing value a feature is having\n\nfig = go.Figure(data=[go.Pie(labels=list(df_20.columns), values=list(df_20.isna().sum()), hole=.3,title='Percentage of Missing values contributed')])\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T09:53:39.319345Z","iopub.execute_input":"2022-08-04T09:53:39.320198Z","iopub.status.idle":"2022-08-04T09:53:39.342454Z","shell.execute_reply.started":"2022-08-04T09:53:39.320142Z","shell.execute_reply":"2022-08-04T09:53:39.340824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we can clearly see nearby 8 features is having more then all over 8 percent of missing value \n# we will select main features only\n\ndf_20 = df_20[['Location','Weekly Cases','Weekly Cases per Million','Weekly Deaths','Weekly Deaths per Million',\"Next Week's Deaths\"]]\ndf_20 = df_20.dropna()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T09:53:39.344861Z","iopub.execute_input":"2022-08-04T09:53:39.346042Z","iopub.status.idle":"2022-08-04T09:53:39.362465Z","shell.execute_reply.started":"2022-08-04T09:53:39.345968Z","shell.execute_reply":"2022-08-04T09:53:39.361058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def func1(df,f,p1,p2):\n    temp = []\n    for i in list(df[f].unique()):\n        temp.append([list(df[f]).count(i),i])\n    temp = sorted(temp,reverse = p1) \n    x = [i[0] for i in temp]\n    y = [i[1] for i in temp]\n    return x[:p2],y[:p2] ","metadata":{"execution":{"iopub.status.busy":"2022-08-04T09:53:39.365126Z","iopub.execute_input":"2022-08-04T09:53:39.365580Z","iopub.status.idle":"2022-08-04T09:53:39.375530Z","shell.execute_reply.started":"2022-08-04T09:53:39.365518Z","shell.execute_reply":"2022-08-04T09:53:39.374017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# with this we will se which location has highest frequency \nx1,y1 = func1(df_20,'Location',True,10)\n\nfig = go.Figure(data=[go.Scatter(\n    x=x1, y=y1,\n    mode='markers',\n    marker=dict(\n        color=[10,20,30,40,50,60,70,80,90,100],\n        size=[10,20,30,40,50,60,70,80,90,100],\n        showscale=True\n        ))\n])\nfig.update_layout(title='Locations with Highest Freq')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T09:53:39.377367Z","iopub.execute_input":"2022-08-04T09:53:39.378733Z","iopub.status.idle":"2022-08-04T09:53:40.140330Z","shell.execute_reply.started":"2022-08-04T09:53:39.378661Z","shell.execute_reply":"2022-08-04T09:53:40.138836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# conclusion we can see world is the highest freq","metadata":{"execution":{"iopub.status.busy":"2022-08-04T09:53:40.145105Z","iopub.execute_input":"2022-08-04T09:53:40.145432Z","iopub.status.idle":"2022-08-04T09:53:40.151324Z","shell.execute_reply.started":"2022-08-04T09:53:40.145403Z","shell.execute_reply":"2022-08-04T09:53:40.149303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# now we will see what locations have maximum weekly cases,weekly Death, next weekly Death\n# because there is no benfit of analysing the locations where there are less covid cases\n\n# maximum\ndf1 = df_20.sort_values(by=['Weekly Cases'],ascending=False)\ndf1 = df1.drop_duplicates(subset='Location',keep='first')\nfig = px.sunburst(df1.head(10),path =list(df1.columns),title=\"TOP 10 Locations with max Weekly Cases\")\nfig.show()\n\ndf1 = df_20.sort_values(by=['Weekly Deaths'],ascending=False)\ndf1 = df1.drop_duplicates(subset='Location',keep='first')\nfig = px.sunburst(df1.head(10),path =list(df1.columns),title=\"TOP 10 Locations with max Weekly Deaths\")\nfig.show()\n\ndf1 = df_20.sort_values(by=[\"Next Week's Deaths\"],ascending=False)\ndf1 = df1.drop_duplicates(subset='Location',keep='first')\nfig = px.sunburst(df1.head(10),path =list(df1.columns),title=\"TOP 10 Locations with max 'Next Week's Deaths'\")\nfig.show()\ndel df1","metadata":{"execution":{"iopub.status.busy":"2022-08-04T10:06:33.872808Z","iopub.execute_input":"2022-08-04T10:06:33.873289Z","iopub.status.idle":"2022-08-04T10:06:34.727538Z","shell.execute_reply.started":"2022-08-04T10:06:33.873241Z","shell.execute_reply":"2022-08-04T10:06:34.725960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_20.columns","metadata":{"execution":{"iopub.status.busy":"2022-08-04T10:03:42.403170Z","iopub.execute_input":"2022-08-04T10:03:42.406078Z","iopub.status.idle":"2022-08-04T10:03:42.420656Z","shell.execute_reply.started":"2022-08-04T10:03:42.405992Z","shell.execute_reply":"2022-08-04T10:03:42.419139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#locations with highest per million cases\n\ndf1 = df_20.sort_values(by=['Weekly Cases per Million'],ascending=False)\ndf1 = df1.drop_duplicates(subset='Location',keep='first')\nfig = px.bar(df1, y='Weekly Cases per Million', x='Location', height=600)\nfig.show()\n\n#locations with highest per million death\ndf1 = df_20.sort_values(by=['Weekly Deaths per Million'],ascending=False)\ndf1 = df1.drop_duplicates(subset='Location',keep='first')\nfig = px.bar(df1, y='Weekly Deaths per Million', x='Location', height=600)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T10:16:59.441327Z","iopub.execute_input":"2022-08-04T10:16:59.441743Z","iopub.status.idle":"2022-08-04T10:16:59.581276Z","shell.execute_reply.started":"2022-08-04T10:16:59.441713Z","shell.execute_reply":"2022-08-04T10:16:59.579832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# more plots coming soon","metadata":{},"execution_count":null,"outputs":[]}]}