{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-03T13:20:31.272053Z","iopub.execute_input":"2022-08-03T13:20:31.272391Z","iopub.status.idle":"2022-08-03T13:20:31.281142Z","shell.execute_reply.started":"2022-08-03T13:20:31.272366Z","shell.execute_reply":"2022-08-03T13:20:31.279972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"../input/titanic/train.csv\")\ndf.head()                 ","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:20:31.282934Z","iopub.execute_input":"2022-08-03T13:20:31.283250Z","iopub.status.idle":"2022-08-03T13:20:31.327147Z","shell.execute_reply.started":"2022-08-03T13:20:31.283218Z","shell.execute_reply":"2022-08-03T13:20:31.326335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:20:31.328103Z","iopub.execute_input":"2022-08-03T13:20:31.328448Z","iopub.status.idle":"2022-08-03T13:20:32.550039Z","shell.execute_reply.started":"2022-08-03T13:20:31.328425Z","shell.execute_reply":"2022-08-03T13:20:32.549079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:20:32.552356Z","iopub.execute_input":"2022-08-03T13:20:32.552627Z","iopub.status.idle":"2022-08-03T13:20:32.581719Z","shell.execute_reply.started":"2022-08-03T13:20:32.552603Z","shell.execute_reply":"2022-08-03T13:20:32.580286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:20:32.583185Z","iopub.execute_input":"2022-08-03T13:20:32.583730Z","iopub.status.idle":"2022-08-03T13:20:32.619204Z","shell.execute_reply.started":"2022-08-03T13:20:32.583661Z","shell.execute_reply":"2022-08-03T13:20:32.618124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_percentage_missing(df:pd.DataFrame,\n                            sort_val:bool = True,\n                            ascending:bool= False,\n                            color:str = \"firebrick\",\n                            col_name = \"percent_missing\",\n                            title:str = \"Percentage of data missing.\",\n                            opacity:float=1.0):\n    if opacity <0 or opacity > 1:\n        raise ValueError(\"Invalid Opacity\")\n    df = get_percentage_missing(df,sort_val,ascending,col_name)\n    fig = px.histogram(df,\n                   x=col_name,\n                   y=df.index,\n                   color_discrete_sequence=[color],\n                   opacity = opacity)\n    fig.update_xaxes(title=\"Percentage (%) Missing\")\n    fig.update_yaxes(title=\"Column Name\")\n    fig.update_layout(title=title)\n    fig.show()\n    \ndef get_percentage_missing(df:pd.DataFrame,sort_val:bool = True,ascending:bool=False,col_name:str=\"percent_missing\"):\n    if sort_val == True :\n        return pd.DataFrame(df.isnull().sum().sort_values(ascending=ascending)/len(df)*100,columns=[col_name])  \n    elif sort_val == False:\n        return pd.DataFrame(df.isnull().sum()/len(df)*100,columns=[col_name])  \n# missing values with percent bar    \nshow_percentage_missing(df=df,ascending=False,sort_val = True,title = \"Percentage of data missing (Train)\",color=\"green\")\n# missing values with barplot\nplt.figure(figsize=(10,6))\nsns.displot(\n    data=df.isna().melt(value_name=\"missing\"),\n    y=\"variable\",\n    hue=\"missing\",\n    multiple=\"fill\",\n    aspect=1.25\n)\nplt.savefig(\"visualizing_missing_data_with_barplot_Seaborn_distplot.png\", dpi=100)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:20:32.621378Z","iopub.execute_input":"2022-08-03T13:20:32.622565Z","iopub.status.idle":"2022-08-03T13:20:34.486899Z","shell.execute_reply.started":"2022-08-03T13:20:32.622524Z","shell.execute_reply":"2022-08-03T13:20:34.485711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# feature violin plot target vs two features (discrete)\nf,ax=plt.subplots(1,2,figsize=(18,8))\nsns.violinplot(x=\"Pclass\", y=\"Age\", hue=\"Survived\", data=df,split=True,ax=ax[0])\nax[0].set_title('Pclass and Age vs Survived')\nax[0].set_yticks(range(0,110,10))\nsns.violinplot(x=\"Sex\", y=\"Age\", hue=\"Survived\", data=df,split=True,ax=ax[1])\nax[1].set_title('Sex and Age vs Survived')\nax[1].set_yticks(range(0,110,10))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:20:34.489109Z","iopub.execute_input":"2022-08-03T13:20:34.490103Z","iopub.status.idle":"2022-08-03T13:20:34.877591Z","shell.execute_reply.started":"2022-08-03T13:20:34.490074Z","shell.execute_reply":"2022-08-03T13:20:34.875904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# correlation heatmap\nsns.heatmap(df.corr(),annot=True,cmap='plasma',linewidths=0.2) #data.corr()-->correlation matrix\nfig=plt.gcf()\nfig.set_size_inches(18,12)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:20:34.879857Z","iopub.execute_input":"2022-08-03T13:20:34.880820Z","iopub.status.idle":"2022-08-03T13:20:35.268076Z","shell.execute_reply.started":"2022-08-03T13:20:34.880792Z","shell.execute_reply":"2022-08-03T13:20:35.267114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import missingno as msno\nmsno.matrix(df)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:20:36.667636Z","iopub.execute_input":"2022-08-03T13:20:36.668524Z","iopub.status.idle":"2022-08-03T13:20:37.054419Z","shell.execute_reply.started":"2022-08-03T13:20:36.668498Z","shell.execute_reply":"2022-08-03T13:20:37.051317Z"},"trusted":true},"execution_count":null,"outputs":[]}]}