{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-13T14:49:41.820097Z","iopub.execute_input":"2022-07-13T14:49:41.820571Z","iopub.status.idle":"2022-07-13T14:49:41.848333Z","shell.execute_reply.started":"2022-07-13T14:49:41.820524Z","shell.execute_reply":"2022-07-13T14:49:41.847408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data=pd.read_csv(\"../input/incomeofindividual/Income-of-individuals.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T14:47:35.591652Z","iopub.execute_input":"2022-07-13T14:47:35.592523Z","iopub.status.idle":"2022-07-13T14:47:35.612575Z","shell.execute_reply.started":"2022-07-13T14:47:35.592480Z","shell.execute_reply":"2022-07-13T14:47:35.611714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T14:47:49.461792Z","iopub.execute_input":"2022-07-13T14:47:49.462256Z","iopub.status.idle":"2022-07-13T14:47:49.503348Z","shell.execute_reply.started":"2022-07-13T14:47:49.462224Z","shell.execute_reply":"2022-07-13T14:47:49.502016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[:10]","metadata":{"execution":{"iopub.status.busy":"2022-07-13T14:47:50.313809Z","iopub.execute_input":"2022-07-13T14:47:50.314291Z","iopub.status.idle":"2022-07-13T14:47:50.345627Z","shell.execute_reply.started":"2022-07-13T14:47:50.314251Z","shell.execute_reply":"2022-07-13T14:47:50.344293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n|Symbol legend:| |\n|-------------|--|\n|A|data quality: excellent|\n|B|data quality: very good|\n|C| data quality: good|","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nprint(\"Shape of Dataset\")\nprint(data.shape)\nprint()\nprint(\"unique elements in Features\")\nprint()\nprint(data.nunique())\nprint()\nprint(\"duplicated Series values\")\nprint(data.duplicated().sum())\nprint()\nprint(\"About Features : \")\nprint()\nprint(data.count()/data.isna().count()*100)\nx=data.count()/data.isna().count()*100\nplt.hist(x)\nplt.ylabel(\"Features of Dataset\")\nplt.xlabel(\"Dataset Present\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T14:47:51.513181Z","iopub.execute_input":"2022-07-13T14:47:51.513744Z","iopub.status.idle":"2022-07-13T14:47:51.822278Z","shell.execute_reply.started":"2022-07-13T14:47:51.513695Z","shell.execute_reply":"2022-07-13T14:47:51.820812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Cleaned Data set**","metadata":{}},{"cell_type":"code","source":"data.to_csv('Income-of-individuals_clean.csv',index=False) #saved","metadata":{"execution":{"iopub.status.busy":"2022-07-13T14:47:52.945135Z","iopub.execute_input":"2022-07-13T14:47:52.945637Z","iopub.status.idle":"2022-07-13T14:47:52.956484Z","shell.execute_reply.started":"2022-07-13T14:47:52.945602Z","shell.execute_reply":"2022-07-13T14:47:52.955030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Exploration of DataSet For Later Use**","metadata":{}},{"cell_type":"code","source":"df=data.T","metadata":{"execution":{"iopub.status.busy":"2022-07-13T14:47:54.663078Z","iopub.execute_input":"2022-07-13T14:47:54.663552Z","iopub.status.idle":"2022-07-13T14:47:54.670601Z","shell.execute_reply.started":"2022-07-13T14:47:54.663515Z","shell.execute_reply":"2022-07-13T14:47:54.669734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.iloc[0:1,1:]","metadata":{"execution":{"iopub.status.busy":"2022-07-13T14:47:55.473022Z","iopub.execute_input":"2022-07-13T14:47:55.473745Z","iopub.status.idle":"2022-07-13T14:47:55.488083Z","shell.execute_reply.started":"2022-07-13T14:47:55.473696Z","shell.execute_reply":"2022-07-13T14:47:55.486961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2 = df.rename(columns={df.columns[0]: 'Geography 4',df.columns[0]: 'Age group',df.columns[1]: 'Sex',df.columns[2]:'Income source',df.columns[3]: 'Statistics',df.columns[4]:'Number of persons (x 1,000)',df.columns[5]: 'Number with income (x 1,000)',df.columns[6]: 'Aggregate income (x 1,000,000)',df.columns[7]: 'Average income (excluding zeros)',df.columns[8]:'Median income (excluding zeros)'})\n","metadata":{"execution":{"iopub.status.busy":"2022-07-13T14:47:56.172949Z","iopub.execute_input":"2022-07-13T14:47:56.173711Z","iopub.status.idle":"2022-07-13T14:47:56.182917Z","shell.execute_reply.started":"2022-07-13T14:47:56.173664Z","shell.execute_reply":"2022-07-13T14:47:56.181533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2","metadata":{"execution":{"iopub.status.busy":"2022-07-13T14:47:56.688179Z","iopub.execute_input":"2022-07-13T14:47:56.688658Z","iopub.status.idle":"2022-07-13T14:47:56.712723Z","shell.execute_reply.started":"2022-07-13T14:47:56.688618Z","shell.execute_reply":"2022-07-13T14:47:56.711194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df3=df2.drop(['Geography 4'], axis=0, inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-13T14:47:57.260230Z","iopub.execute_input":"2022-07-13T14:47:57.261369Z","iopub.status.idle":"2022-07-13T14:47:57.271066Z","shell.execute_reply.started":"2022-07-13T14:47:57.261320Z","shell.execute_reply":"2022-07-13T14:47:57.269587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2","metadata":{"execution":{"iopub.status.busy":"2022-07-13T14:47:57.752854Z","iopub.execute_input":"2022-07-13T14:47:57.753290Z","iopub.status.idle":"2022-07-13T14:47:57.779125Z","shell.execute_reply.started":"2022-07-13T14:47:57.753258Z","shell.execute_reply":"2022-07-13T14:47:57.777708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#'Geography 4','Age group','Sex','Income source','Statistics','Number of persons (x 1,000)','Number with income (x 1,000)','Aggregate income (x 1,000,000)','Average income (excluding zeros)','Median income (excluding zeros)'","metadata":{"execution":{"iopub.status.busy":"2022-07-13T14:47:58.472362Z","iopub.execute_input":"2022-07-13T14:47:58.474080Z","iopub.status.idle":"2022-07-13T14:47:58.479247Z","shell.execute_reply.started":"2022-07-13T14:47:58.474018Z","shell.execute_reply":"2022-07-13T14:47:58.477937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2['Geography 4'] = df2.index\ndf3=df2.reset_index()\ndf3.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T14:47:59.085923Z","iopub.execute_input":"2022-07-13T14:47:59.087314Z","iopub.status.idle":"2022-07-13T14:47:59.108188Z","shell.execute_reply.started":"2022-07-13T14:47:59.087261Z","shell.execute_reply":"2022-07-13T14:47:59.106960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df4=df3.drop(['index'], axis = 1)\ndf4","metadata":{"execution":{"iopub.status.busy":"2022-07-13T14:47:59.860759Z","iopub.execute_input":"2022-07-13T14:47:59.861513Z","iopub.status.idle":"2022-07-13T14:47:59.885611Z","shell.execute_reply.started":"2022-07-13T14:47:59.861453Z","shell.execute_reply":"2022-07-13T14:47:59.884398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px","metadata":{"execution":{"iopub.status.busy":"2022-07-13T14:48:00.566159Z","iopub.execute_input":"2022-07-13T14:48:00.566705Z","iopub.status.idle":"2022-07-13T14:48:02.145986Z","shell.execute_reply.started":"2022-07-13T14:48:00.566654Z","shell.execute_reply":"2022-07-13T14:48:02.144369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.scatter(df4, x=\"Geography 4\", y=\"Number of persons (x 1,000)\",color='Statistics')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T14:48:02.149179Z","iopub.execute_input":"2022-07-13T14:48:02.150057Z","iopub.status.idle":"2022-07-13T14:48:03.299893Z","shell.execute_reply.started":"2022-07-13T14:48:02.150011Z","shell.execute_reply":"2022-07-13T14:48:03.298695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.scatter(df4, x=\"Geography 4\", y=\"Number with income (x 1,000)\",color='Statistics')\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-13T14:48:03.301478Z","iopub.execute_input":"2022-07-13T14:48:03.301850Z","iopub.status.idle":"2022-07-13T14:48:03.389892Z","shell.execute_reply.started":"2022-07-13T14:48:03.301817Z","shell.execute_reply":"2022-07-13T14:48:03.388716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df4.to_csv('Income-of-individuals_eda.csv',index=False) #saved data ready for eda","metadata":{"execution":{"iopub.status.busy":"2022-07-13T14:48:06.706780Z","iopub.execute_input":"2022-07-13T14:48:06.708145Z","iopub.status.idle":"2022-07-13T14:48:06.716405Z","shell.execute_reply.started":"2022-07-13T14:48:06.708093Z","shell.execute_reply":"2022-07-13T14:48:06.715104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_1999=pd.read_csv(\"../input/incomeofindividuals19992020/Income-of-individuals1999-2020.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T14:50:06.087993Z","iopub.execute_input":"2022-07-13T14:50:06.088455Z","iopub.status.idle":"2022-07-13T14:50:06.141374Z","shell.execute_reply.started":"2022-07-13T14:50:06.088418Z","shell.execute_reply":"2022-07-13T14:50:06.140012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_1999.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T14:50:16.879415Z","iopub.execute_input":"2022-07-13T14:50:16.879896Z","iopub.status.idle":"2022-07-13T14:50:16.908330Z","shell.execute_reply.started":"2022-07-13T14:50:16.879860Z","shell.execute_reply":"2022-07-13T14:50:16.907002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_1999.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T14:50:26.577798Z","iopub.execute_input":"2022-07-13T14:50:26.578242Z","iopub.status.idle":"2022-07-13T14:50:27.311545Z","shell.execute_reply.started":"2022-07-13T14:50:26.578207Z","shell.execute_reply":"2022-07-13T14:50:27.310403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Shape of Dataset\")\nprint(df_1999.shape)\nprint()\nprint(\"unique elements in Features\")\nprint()\nprint(df_1999.nunique())\nprint()\nprint(\"duplicated Series values\")\nprint(df_1999.duplicated().sum())\nprint()\nprint(\"About Features : \")\nprint()\nprint(df_1999.count()/df_1999.isna().count()*100)\nx=df_1999.count()/df_1999.isna().count()*100\nplt.hist(x)\nplt.ylabel(\"Features of Dataset\")\nplt.xlabel(\"Dataset Present\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T14:50:46.900699Z","iopub.execute_input":"2022-07-13T14:50:46.901172Z","iopub.status.idle":"2022-07-13T14:50:47.244283Z","shell.execute_reply.started":"2022-07-13T14:50:46.901137Z","shell.execute_reply":"2022-07-13T14:50:47.242683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_1999.to_csv('Income-of-individuals_1999to2020.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T14:51:35.485700Z","iopub.execute_input":"2022-07-13T14:51:35.486644Z","iopub.status.idle":"2022-07-13T14:51:35.502954Z","shell.execute_reply.started":"2022-07-13T14:51:35.486592Z","shell.execute_reply":"2022-07-13T14:51:35.501392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}