{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-29T04:51:26.715727Z","iopub.execute_input":"2022-07-29T04:51:26.716200Z","iopub.status.idle":"2022-07-29T04:51:26.725826Z","shell.execute_reply.started":"2022-07-29T04:51:26.716154Z","shell.execute_reply":"2022-07-29T04:51:26.724419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Hi! This is my first attempt to present my work to the Kaggle community. Any feedback is appreciated. Thank you!","metadata":{}},{"cell_type":"markdown","source":"# Exploratory Data Analysis","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/udacity-mlcharity-competition/census.csv\")\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:26.727318Z","iopub.execute_input":"2022-07-29T04:51:26.727646Z","iopub.status.idle":"2022-07-29T04:51:26.862800Z","shell.execute_reply.started":"2022-07-29T04:51:26.727608Z","shell.execute_reply":"2022-07-29T04:51:26.861682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:26.864724Z","iopub.execute_input":"2022-07-29T04:51:26.865107Z","iopub.status.idle":"2022-07-29T04:51:26.926457Z","shell.execute_reply.started":"2022-07-29T04:51:26.865073Z","shell.execute_reply":"2022-07-29T04:51:26.925479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv(\"/kaggle/input/udacity-mlcharity-competition/test_census.csv\")\ndf_test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:26.927923Z","iopub.execute_input":"2022-07-29T04:51:26.928288Z","iopub.status.idle":"2022-07-29T04:51:27.090539Z","shell.execute_reply.started":"2022-07-29T04:51:26.928255Z","shell.execute_reply":"2022-07-29T04:51:27.089355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"No null values in the training dataset. However there are some null values in the testing dataset. While training will not be affected, testing might need to fill with median or most common categorical values.","metadata":{}},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:27.092192Z","iopub.execute_input":"2022-07-29T04:51:27.093079Z","iopub.status.idle":"2022-07-29T04:51:27.129064Z","shell.execute_reply.started":"2022-07-29T04:51:27.093040Z","shell.execute_reply":"2022-07-29T04:51:27.128182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.drop(\"Unnamed: 0\", axis=1).describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:27.130369Z","iopub.execute_input":"2022-07-29T04:51:27.131300Z","iopub.status.idle":"2022-07-29T04:51:27.175789Z","shell.execute_reply.started":"2022-07-29T04:51:27.131262Z","shell.execute_reply":"2022-07-29T04:51:27.174448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The capital columns seem very susceptible to outliers. Overall distributions in the training dataset and testing dataset seem pretty similar.","metadata":{}},{"cell_type":"markdown","source":"Now to analyse column by column to gain insights.","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nfrom matplotlib.lines import Line2D\nfrom matplotlib.patches import Patch\nfrom sklearn.feature_selection import chi2\nfrom sklearn.preprocessing import PowerTransformer\n# from sklearn.feature_selection import mutual_info_classif\nfrom scipy import stats","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:27.177207Z","iopub.execute_input":"2022-07-29T04:51:27.177673Z","iopub.status.idle":"2022-07-29T04:51:27.184797Z","shell.execute_reply.started":"2022-07-29T04:51:27.177627Z","shell.execute_reply":"2022-07-29T04:51:27.183630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def analyse_column(col):\n    \"\"\"\n    Given column name plot distributions depending on whether the column is a \n    categorical or a numeric feature. If categorical, collate minorities into\n    an \"others\" category if > 11 unique categories.\n    \"\"\"\n    if df[col].dtype == 'object':\n        nunique = df[col].nunique()\n        print(f\"{col} - {nunique} unique values.\")\n        df_ref = df.groupby([col, \"income\"]).size().unstack()\n        df_ref[\"Total\"] = df_ref.sum(axis=1)\n        df_ref = df_ref.sort_values(\"Total\", ascending=False)\n        if nunique > 11:\n            print(f\"Top 10 categories: {', '.join(df_ref.iloc[:10].index)}\")\n            print(f\"Others categories: {', '.join(df_ref.iloc[10:].index)}.\")\n            df_others = pd.DataFrame(df_ref.iloc[10:].sum(), columns=[\"Others\"]).T\n            df_ref = pd.concat([df_ref[:10], df_others])\n        else:\n            print(f\"All categories: {', '.join(df_ref.iloc[:10].index)}\")\n        df_ref = df_ref / df_ref.sum()\n        fig, ax = plt.subplots()\n        legend_elements = [Line2D([0], [0], color=\"purple\", lw=3, label=\"All\"),\n                           Patch(facecolor=\"red\", alpha=0.5, label=\">50K\"),\n                           Patch(facecolor=\"blue\", alpha=0.5, label=\"<=50K\")]\n        fig = sns.lineplot(data=df_ref, x=df_ref.index, y=\"Total\", color=\"purple\")\n        fig = sns.barplot(data=df_ref, x=df_ref.index, y=\">50K\", color=\"red\", alpha=0.5)\n        fig = sns.barplot(data=df_ref, x=df_ref.index, y=\"<=50K\", color=\"blue\", alpha=0.5)\n        plt.ylabel(\"Density\")\n        plt.xticks(rotation=90)\n        ax.legend(handles=legend_elements)\n        plt.show()\n    else:\n        print(f\"{col} - mean: {df[col].mean():.2f} median: {df[col].median()}\")\n        fig, ax = plt.subplots()\n        legend_elements = [Line2D([0], [0], color=\"grey\", lw=3, label=\"All\"),\n                           Line2D([0], [0], color=\"red\", lw=3, label=\">50K\"),\n                           Line2D([0], [0], color=\"blue\", lw=3, label=\"<=50K\")]\n        fig = sns.kdeplot(df[col], color=\"grey\", shade=True)\n        fig = sns.kdeplot(df[df[\"income\"] == \">50K\"][col], color=\"red\")\n        fig = sns.kdeplot(df[df[\"income\"] == \"<=50K\"][col], color=\"blue\")\n        ax.legend(handles=legend_elements)\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:27.186156Z","iopub.execute_input":"2022-07-29T04:51:27.186536Z","iopub.status.idle":"2022-07-29T04:51:27.207041Z","shell.execute_reply.started":"2022-07-29T04:51:27.186502Z","shell.execute_reply":"2022-07-29T04:51:27.205806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def categorical_chi2(column, print_crosstab=True):\n    \"\"\"\n    Tallies categories of each income type and performs a chi2 test on whether the\n    two distributions are from the same distribution.\n    \"\"\"\n    df_crosstab = pd.crosstab(df[column], df[\"income\"], margins=True)\n    if print_crosstab:\n        print(df_crosstab)\n    results = stats.chi2_contingency(df_crosstab.to_numpy())[:-1]\n    print(f\"chi2: {results[0]}\")\n    print(f\"p_value: {results[1]}\")\n    print(f\"DOF: {results[2]}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:27.208395Z","iopub.execute_input":"2022-07-29T04:51:27.209542Z","iopub.status.idle":"2022-07-29T04:51:27.223764Z","shell.execute_reply.started":"2022-07-29T04:51:27.209494Z","shell.execute_reply":"2022-07-29T04:51:27.222614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Age","metadata":{}},{"cell_type":"code","source":"analyse_column(\"age\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:27.226647Z","iopub.execute_input":"2022-07-29T04:51:27.227067Z","iopub.status.idle":"2022-07-29T04:51:27.936610Z","shell.execute_reply.started":"2022-07-29T04:51:27.227032Z","shell.execute_reply":"2022-07-29T04:51:27.935449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stats.ks_2samp(df[df[\"income\"] == \">50K\"][\"age\"], df[df[\"income\"] == \"<=50K\"][\"age\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:27.937967Z","iopub.execute_input":"2022-07-29T04:51:27.938333Z","iopub.status.idle":"2022-07-29T04:51:27.978221Z","shell.execute_reply.started":"2022-07-29T04:51:27.938298Z","shell.execute_reply":"2022-07-29T04:51:27.977150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Evidently the distributions for different income groups are different. Age is a very important feature for this dataset. However the curves are slightly skewed so might need some preprocessing.","metadata":{}},{"cell_type":"code","source":"df['age'].values.reshape(1, -1)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:27.979902Z","iopub.execute_input":"2022-07-29T04:51:27.980646Z","iopub.status.idle":"2022-07-29T04:51:27.988386Z","shell.execute_reply.started":"2022-07-29T04:51:27.980600Z","shell.execute_reply":"2022-07-29T04:51:27.987448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Workclass","metadata":{}},{"cell_type":"code","source":"analyse_column(\"workclass\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:27.990080Z","iopub.execute_input":"2022-07-29T04:51:27.990798Z","iopub.status.idle":"2022-07-29T04:51:28.296296Z","shell.execute_reply.started":"2022-07-29T04:51:27.990753Z","shell.execute_reply":"2022-07-29T04:51:28.294961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_chi2(\"workclass\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:28.299024Z","iopub.execute_input":"2022-07-29T04:51:28.299496Z","iopub.status.idle":"2022-07-29T04:51:28.379495Z","shell.execute_reply.started":"2022-07-29T04:51:28.299447Z","shell.execute_reply":"2022-07-29T04:51:28.378328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Education","metadata":{}},{"cell_type":"markdown","source":"Private workclass tends to have lower income than other workclasses.","metadata":{}},{"cell_type":"code","source":"analyse_column(\"education_level\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:28.380868Z","iopub.execute_input":"2022-07-29T04:51:28.381489Z","iopub.status.idle":"2022-07-29T04:51:28.914118Z","shell.execute_reply.started":"2022-07-29T04:51:28.381454Z","shell.execute_reply":"2022-07-29T04:51:28.912990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_chi2(\"education_level\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:28.915524Z","iopub.execute_input":"2022-07-29T04:51:28.915871Z","iopub.status.idle":"2022-07-29T04:51:28.995831Z","shell.execute_reply.started":"2022-07-29T04:51:28.915838Z","shell.execute_reply":"2022-07-29T04:51:28.994507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Bachelors, Masters and Prof-school education levels have significantly higher distributions of high income reports.","metadata":{}},{"cell_type":"code","source":"analyse_column(\"education-num\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:28.997592Z","iopub.execute_input":"2022-07-29T04:51:28.998230Z","iopub.status.idle":"2022-07-29T04:51:29.699788Z","shell.execute_reply.started":"2022-07-29T04:51:28.998172Z","shell.execute_reply":"2022-07-29T04:51:29.698694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It is unclear what the education number means. From the graph it can be related to the estimated total length of education.","metadata":{}},{"cell_type":"code","source":"pd.crosstab(df[\"education_level\"], df[\"education-num\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:29.701385Z","iopub.execute_input":"2022-07-29T04:51:29.701737Z","iopub.status.idle":"2022-07-29T04:51:29.749200Z","shell.execute_reply.started":"2022-07-29T04:51:29.701704Z","shell.execute_reply":"2022-07-29T04:51:29.748138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It seems to be match each category in the education level column. So only one of the columns should be retained.","metadata":{}},{"cell_type":"markdown","source":"## Marital Status","metadata":{}},{"cell_type":"code","source":"analyse_column(\"marital-status\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:29.751105Z","iopub.execute_input":"2022-07-29T04:51:29.751451Z","iopub.status.idle":"2022-07-29T04:51:30.054075Z","shell.execute_reply.started":"2022-07-29T04:51:29.751420Z","shell.execute_reply":"2022-07-29T04:51:30.052980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_chi2(\"marital-status\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:30.055389Z","iopub.execute_input":"2022-07-29T04:51:30.055708Z","iopub.status.idle":"2022-07-29T04:51:30.134181Z","shell.execute_reply.started":"2022-07-29T04:51:30.055677Z","shell.execute_reply":"2022-07-29T04:51:30.132981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"While it is statistically significant that marital status has a huge income, it is suspected that it is due to the age distribution. Now to plot marital status against age:","metadata":{}},{"cell_type":"code","source":"fig = sns.boxplot(data=df, x=\"marital-status\", y=\"age\", hue=\"income\")\nplt.xticks(rotation=90)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:30.137179Z","iopub.execute_input":"2022-07-29T04:51:30.137638Z","iopub.status.idle":"2022-07-29T04:51:30.898770Z","shell.execute_reply.started":"2022-07-29T04:51:30.137593Z","shell.execute_reply":"2022-07-29T04:51:30.897414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It seems there is a correlation but not entirely conclusive.","metadata":{}},{"cell_type":"markdown","source":"## Occupation","metadata":{}},{"cell_type":"code","source":"analyse_column(\"occupation\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:30.900797Z","iopub.execute_input":"2022-07-29T04:51:30.901327Z","iopub.status.idle":"2022-07-29T04:51:31.248033Z","shell.execute_reply.started":"2022-07-29T04:51:30.901262Z","shell.execute_reply":"2022-07-29T04:51:31.246744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_chi2(\"occupation\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:31.250069Z","iopub.execute_input":"2022-07-29T04:51:31.250537Z","iopub.status.idle":"2022-07-29T04:51:31.332134Z","shell.execute_reply.started":"2022-07-29T04:51:31.250490Z","shell.execute_reply":"2022-07-29T04:51:31.330997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The occupation categories of Prof-specialty,  Exec-managerial seem to very informative. Now to plot their age distributions to see if they are correlated.","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:53:03.446058Z","iopub.execute_input":"2022-07-26T00:53:03.446434Z","iopub.status.idle":"2022-07-26T00:53:03.453413Z","shell.execute_reply.started":"2022-07-26T00:53:03.446405Z","shell.execute_reply":"2022-07-26T00:53:03.452002Z"}}},{"cell_type":"code","source":"fig, ax = plt.subplots()\nlegend_elements = [Line2D([0], [0], color=\"grey\", lw=3, label=\"All\"),\n                   Line2D([0], [0], color=\"red\", lw=3, label=\"Prof-specialty\"),\n                   Line2D([0], [0], color=\"blue\", lw=3, label=\"Exec-managerial\"),\n                   Line2D([0], [0], color=\"green\", lw=3, label=\"Others\")]\nfig = sns.kdeplot(df[\"age\"], color=\"grey\", shade=True)\nfig = sns.kdeplot(df[df[\"occupation\"] == \" Prof-specialty\"][\"age\"], color=\"red\")\nfig = sns.kdeplot(df[df[\"occupation\"] == \" Exec-managerial\"][\"age\"], color=\"blue\")\nfig = sns.kdeplot(df[~df[\"occupation\"].isin([\" Prof-specialty\", \" Exec-managerial\"])][\"age\"], color=\"green\")\nax.legend(handles=legend_elements)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:31.335084Z","iopub.execute_input":"2022-07-29T04:51:31.336286Z","iopub.status.idle":"2022-07-29T04:51:32.019329Z","shell.execute_reply.started":"2022-07-29T04:51:31.336236Z","shell.execute_reply":"2022-07-29T04:51:32.018009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Their age distributions seem significantly higher than average.","metadata":{}},{"cell_type":"code","source":"stats.ks_2samp(df[df[\"occupation\"] == \" Prof-specialty\"][\"age\"], df[df[\"occupation\"] != \" Prof-specialty\"][\"age\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:32.020856Z","iopub.execute_input":"2022-07-29T04:51:32.021397Z","iopub.status.idle":"2022-07-29T04:51:32.067482Z","shell.execute_reply.started":"2022-07-29T04:51:32.021354Z","shell.execute_reply":"2022-07-29T04:51:32.066343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stats.ks_2samp(df[df[\"occupation\"] == \" Exec-managerial\"][\"age\"], df[df[\"occupation\"] != \" Exec-managerial\"][\"age\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:32.069266Z","iopub.execute_input":"2022-07-29T04:51:32.070075Z","iopub.status.idle":"2022-07-29T04:51:32.116718Z","shell.execute_reply.started":"2022-07-29T04:51:32.070027Z","shell.execute_reply":"2022-07-29T04:51:32.115600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"While the results are statistically significant the difference is small. Now to check their relationship with workclass.","metadata":{}},{"cell_type":"code","source":"pd.crosstab(df[\"workclass\"], df[\"occupation\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:32.118442Z","iopub.execute_input":"2022-07-29T04:51:32.119139Z","iopub.status.idle":"2022-07-29T04:51:32.166058Z","shell.execute_reply.started":"2022-07-29T04:51:32.119092Z","shell.execute_reply":"2022-07-29T04:51:32.164837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"While there are some occupations that are entirely in 1 workclass, this table does not seem very informative on a whole.","metadata":{}},{"cell_type":"markdown","source":"## Relationship","metadata":{}},{"cell_type":"code","source":"analyse_column(\"relationship\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:32.168017Z","iopub.execute_input":"2022-07-29T04:51:32.168837Z","iopub.status.idle":"2022-07-29T04:51:32.451837Z","shell.execute_reply.started":"2022-07-29T04:51:32.168774Z","shell.execute_reply":"2022-07-29T04:51:32.450546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.crosstab(df[\"relationship\"], df[\"marital-status\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:32.453768Z","iopub.execute_input":"2022-07-29T04:51:32.454276Z","iopub.status.idle":"2022-07-29T04:51:32.496077Z","shell.execute_reply.started":"2022-07-29T04:51:32.454229Z","shell.execute_reply":"2022-07-29T04:51:32.494990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It seems slightly different from the information in marital-status column. But still *Husband* and *Wife* are predominantly in *Married-civ-spouse* and they all have high distributions of high incomes.","metadata":{}},{"cell_type":"markdown","source":"## Race","metadata":{}},{"cell_type":"code","source":"analyse_column(\"race\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:32.497734Z","iopub.execute_input":"2022-07-29T04:51:32.498096Z","iopub.status.idle":"2022-07-29T04:51:32.748483Z","shell.execute_reply.started":"2022-07-29T04:51:32.498058Z","shell.execute_reply":"2022-07-29T04:51:32.747368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots()\nlegend_elements = [Patch(facecolor=\"red\", alpha=0.5, label=\"Black Population\"),\n                   Patch(facecolor=\"grey\", alpha=0.5, label=\"Whole Population\")]\nfig = sns.histplot(data=df[df[\"race\"] == \" Black\"][\"occupation\"], stat=\"density\", color=\"red\", alpha=0.5)\nfig = sns.histplot(data=df[\"occupation\"], stat=\"density\", color=\"grey\", alpha=0.5)\nplt.xticks(rotation=90)\nax.legend(handles=legend_elements)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:32.749789Z","iopub.execute_input":"2022-07-29T04:51:32.750285Z","iopub.status.idle":"2022-07-29T04:51:33.201873Z","shell.execute_reply.started":"2022-07-29T04:51:32.750250Z","shell.execute_reply":"2022-07-29T04:51:33.201018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It seems high paying occupations such as Prof-specialty, Exec-managerial and Sales have significantly lower black population.","metadata":{}},{"cell_type":"markdown","source":"## Sex","metadata":{}},{"cell_type":"code","source":"analyse_column(\"sex\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:33.203469Z","iopub.execute_input":"2022-07-29T04:51:33.204633Z","iopub.status.idle":"2022-07-29T04:51:33.417860Z","shell.execute_reply.started":"2022-07-29T04:51:33.204584Z","shell.execute_reply":"2022-07-29T04:51:33.416548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"While it seems male earn more than female, the amount of female records are quite low compared to male records. Let us check their age distribution.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots()\nlegend_elements = [Line2D([0], [0], color=\"grey\", lw=3, label=\"All\"),\n                   Line2D([0], [0], color=\"red\", lw=3, label=\"Female\"),\n                   Line2D([0], [0], color=\"blue\", lw=3, label=\"Male\")]\nfig = sns.kdeplot(df[\"age\"], color=\"grey\", shade=True)\nfig = sns.kdeplot(df[df[\"sex\"] == \" Female\"][\"age\"], color=\"red\")\nfig = sns.kdeplot(df[df[\"sex\"] == \" Male\"][\"age\"], color=\"blue\")\nax.legend(handles=legend_elements)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:33.419244Z","iopub.execute_input":"2022-07-29T04:51:33.419578Z","iopub.status.idle":"2022-07-29T04:51:34.139004Z","shell.execute_reply.started":"2022-07-29T04:51:33.419547Z","shell.execute_reply":"2022-07-29T04:51:34.137719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The average age of female records is lower than that of male records. That might contribute to the income distribution.","metadata":{}},{"cell_type":"markdown","source":"## Capital","metadata":{}},{"cell_type":"code","source":"analyse_column(\"capital-gain\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:34.140620Z","iopub.execute_input":"2022-07-29T04:51:34.141450Z","iopub.status.idle":"2022-07-29T04:51:34.920906Z","shell.execute_reply.started":"2022-07-29T04:51:34.141402Z","shell.execute_reply":"2022-07-29T04:51:34.919622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This graph is not informative since it is dominated by 0.","metadata":{}},{"cell_type":"code","source":"df_nonzero = df[df[\"capital-gain\"] != 0]\ndf_nonzero.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:34.922560Z","iopub.execute_input":"2022-07-29T04:51:34.923031Z","iopub.status.idle":"2022-07-29T04:51:34.958104Z","shell.execute_reply.started":"2022-07-29T04:51:34.922983Z","shell.execute_reply":"2022-07-29T04:51:34.956921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_nonzero[\"income\"].hist()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:34.959815Z","iopub.execute_input":"2022-07-29T04:51:34.960816Z","iopub.status.idle":"2022-07-29T04:51:35.132288Z","shell.execute_reply.started":"2022-07-29T04:51:34.960778Z","shell.execute_reply":"2022-07-29T04:51:35.131134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots()\nlegend_elements = [Line2D([0], [0], color=\"grey\", lw=3, label=\"All\"),\n                   Line2D([0], [0], color=\"red\", lw=3, label=\">50K\"),\n                   Line2D([0], [0], color=\"blue\", lw=3, label=\"<=50K\")]\nfig = sns.kdeplot(df_nonzero[\"capital-gain\"], color=\"grey\", shade=True)\nfig = sns.kdeplot(df_nonzero[df_nonzero[\"income\"] == \">50K\"][\"capital-gain\"], color=\"red\")\nfig = sns.kdeplot(df_nonzero[df_nonzero[\"income\"] == \"<=50K\"][\"capital-gain\"], color=\"blue\")\nax.legend(handles=legend_elements)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:35.133544Z","iopub.execute_input":"2022-07-29T04:51:35.133869Z","iopub.status.idle":"2022-07-29T04:51:35.438086Z","shell.execute_reply.started":"2022-07-29T04:51:35.133839Z","shell.execute_reply":"2022-07-29T04:51:35.436995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"High income people tend to have high gain.","metadata":{}},{"cell_type":"code","source":"stats.ks_2samp(df[df[\"income\"] == \">50K\"][\"capital-gain\"], df[df[\"income\"] == \"<=50K\"][\"capital-gain\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:35.439691Z","iopub.execute_input":"2022-07-29T04:51:35.440255Z","iopub.status.idle":"2022-07-29T04:51:35.485968Z","shell.execute_reply.started":"2022-07-29T04:51:35.440220Z","shell.execute_reply":"2022-07-29T04:51:35.484924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It seems a fairly good predictor of income.","metadata":{}},{"cell_type":"code","source":"analyse_column(\"capital-loss\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:35.487509Z","iopub.execute_input":"2022-07-29T04:51:35.488192Z","iopub.status.idle":"2022-07-29T04:51:36.441532Z","shell.execute_reply.started":"2022-07-29T04:51:35.488156Z","shell.execute_reply":"2022-07-29T04:51:36.440582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stats.ks_2samp(df[df[\"income\"] == \">50K\"][\"capital-loss\"], df[df[\"income\"] == \"<=50K\"][\"capital-loss\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:36.442717Z","iopub.execute_input":"2022-07-29T04:51:36.443539Z","iopub.status.idle":"2022-07-29T04:51:36.493821Z","shell.execute_reply.started":"2022-07-29T04:51:36.443503Z","shell.execute_reply":"2022-07-29T04:51:36.492670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The distribution graph looks less extreme than the gain graph. However the difference in distribution is smaller compared to gain.\n\nDue to 0 dominating the capital columns and the presence of extreme values. It might be better to convert these two into categorical columns for whether having a non-zero value.","metadata":{}},{"cell_type":"markdown","source":"## Hours Per Week","metadata":{}},{"cell_type":"code","source":"analyse_column(\"hours-per-week\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:36.495286Z","iopub.execute_input":"2022-07-29T04:51:36.495716Z","iopub.status.idle":"2022-07-29T04:51:37.253237Z","shell.execute_reply.started":"2022-07-29T04:51:36.495684Z","shell.execute_reply":"2022-07-29T04:51:37.251965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stats.ks_2samp(df[df[\"income\"] == \">50K\"][\"hours-per-week\"], df[df[\"income\"] == \"<=50K\"][\"hours-per-week\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:37.254607Z","iopub.execute_input":"2022-07-29T04:51:37.254936Z","iopub.status.idle":"2022-07-29T04:51:37.295048Z","shell.execute_reply.started":"2022-07-29T04:51:37.254906Z","shell.execute_reply":"2022-07-29T04:51:37.293968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"High income people tend to work longer hours, which makes sense.","metadata":{}},{"cell_type":"markdown","source":"## Native Country","metadata":{}},{"cell_type":"code","source":"analyse_column(\"native-country\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:37.296391Z","iopub.execute_input":"2022-07-29T04:51:37.296709Z","iopub.status.idle":"2022-07-29T04:51:37.672363Z","shell.execute_reply.started":"2022-07-29T04:51:37.296679Z","shell.execute_reply":"2022-07-29T04:51:37.671122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_chi2(\"native-country\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:37.673764Z","iopub.execute_input":"2022-07-29T04:51:37.674126Z","iopub.status.idle":"2022-07-29T04:51:37.753993Z","shell.execute_reply.started":"2022-07-29T04:51:37.674092Z","shell.execute_reply":"2022-07-29T04:51:37.752729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The column seems informative with US natives having higher income while Mexicans having lower income from the distribution graph. However there are too many unique categories and thus might need to be pruned before feeding into the model.","metadata":{}},{"cell_type":"markdown","source":"# Preprocessing\n\nBelow are referred from Jose Guzman's [work](https://www.kaggle.com/code/joseguzman/8-models-that-predict-autism-spectrum-disorder) on Autism Prediction competition.","metadata":{}},{"cell_type":"code","source":"from sklearn.base import BaseEstimator, TransformerMixin\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.ensemble import ExtraTreesClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import roc_curve\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.metrics import RocCurveDisplay\nfrom imblearn.combine import SMOTEENN\nfrom imblearn.under_sampling import EditedNearestNeighbours","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:37.755690Z","iopub.execute_input":"2022-07-29T04:51:37.756049Z","iopub.status.idle":"2022-07-29T04:51:37.767050Z","shell.execute_reply.started":"2022-07-29T04:51:37.756016Z","shell.execute_reply":"2022-07-29T04:51:37.765801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CapitalTransformer(BaseEstimator, TransformerMixin):\n    \"\"\"\n    A custom feature transformer to convert the two capital columns into 0 and 1.\n    \"\"\"\n    def __init__(self):\n        self.df = None\n    \n    def get_feature_names_out(self, input_features=None):\n        if self.df is None:\n            mycols = ['None']\n        else:\n            mycols =  self.df.columns.tolist()\n            \n        return mycols\n\n    def fit(self, X:pd.DataFrame, y = None):\n        self.df = X.copy()\n        \n        return self\n   \n    def transform(self, X:pd.DataFrame = None):\n        df = X.copy()\n        df[\"capital-gain\"] = (df[\"capital-gain\"] > 0).astype(int)\n        df[\"capital-loss\"] = (df[\"capital-loss\"] > 0).astype(int)\n        \n        return df","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:37.768695Z","iopub.execute_input":"2022-07-29T04:51:37.769233Z","iopub.status.idle":"2022-07-29T04:51:37.780497Z","shell.execute_reply.started":"2022-07-29T04:51:37.769197Z","shell.execute_reply":"2022-07-29T04:51:37.779279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CountryTransformer(BaseEstimator, TransformerMixin):\n    \"\"\"\n    A custom feature transformer to prune native-country into 3 categories: US, Mexico and Others.\n    \"\"\"\n    def __init__(self):\n        self.df = None\n    \n    def get_feature_names_out(self, input_features=None):\n        if self.df is None:\n            mycols = ['None']\n        else:\n            mycols =  self.df.columns.tolist()\n            \n        return mycols\n\n    def fit(self, X:pd.DataFrame, y = None):\n        self.df = X.copy()\n        \n        return self\n   \n    def transform(self, X:pd.DataFrame = None):\n        df = X.copy()\n        subset = df[~df[\"native-country\"].isin([\"  United-States\", \"Mexico\"])]\n        df.loc[subset.index, \"native-country\"] = \"Others\"\n        \n        return df","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:37.781990Z","iopub.execute_input":"2022-07-29T04:51:37.783031Z","iopub.status.idle":"2022-07-29T04:51:37.792349Z","shell.execute_reply.started":"2022-07-29T04:51:37.782996Z","shell.execute_reply":"2022-07-29T04:51:37.791266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class EducationTransformer(BaseEstimator, TransformerMixin):\n    \"\"\"\n    A custom feature transformer to drop the education-level column.\n    \"\"\"\n    def __init__(self):\n        self.df = None\n    \n    def get_feature_names_out(self, input_features=None):\n        if self.df is None:\n            mycols = ['None']\n        else:\n            mycols =  self.df.columns.tolist()\n            \n        if \"education_level\" in mycols:\n            mycols.remove(\"education_level\")\n            \n        return mycols\n\n    def fit(self, X:pd.DataFrame, y = None):\n        self.df = X.copy()\n        \n        return self\n   \n    def transform(self, X:pd.DataFrame = None):\n        df = X.copy()\n        df = df.drop(\"education_level\", axis=1)\n        \n        return df","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:37.794027Z","iopub.execute_input":"2022-07-29T04:51:37.794740Z","iopub.status.idle":"2022-07-29T04:51:37.807990Z","shell.execute_reply.started":"2022-07-29T04:51:37.794706Z","shell.execute_reply":"2022-07-29T04:51:37.806897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class MyPreprocess():\n    \"\"\"\n    A replacement to pipeline due to ColumnTransformer converting pandas DataFrame to numpy Array,\n    the second ColumnTransformer would not be able to recognise column names. Therefore, a step to\n    reconvert them into pandas DataFrame.\n    \"\"\"\n    def __init__(self):\n        self.preprocess1 = ColumnTransformer(transformers=(\n                (\"capital\", CapitalTransformer(), [\"capital-gain\", \"capital-loss\"]),\n                (\"country\", CountryTransformer(), [\"native-country\"]),\n                (\"education\", EducationTransformer(), [\"education_level\", \"education-num\"]),\n            ), remainder=\"passthrough\", verbose_feature_names_out=False)\n        self.preprocess2 = ColumnTransformer(transformers=(\n                (\"binary\", OneHotEncoder(sparse=False, drop= \"if_binary\"), [\"sex\", \"capital-gain\", \"capital-loss\"]),\n                (\"categorical\", OneHotEncoder(sparse=False), [\"workclass\", \"marital-status\", \"occupation\", \"relationship\", \"race\", \"native-country\"]),\n                (\"z-score\", StandardScaler(), [\"age\", \"education-num\", \"hours-per-week\"]),\n            ), remainder=\"passthrough\", verbose_feature_names_out=False)\n        self.sm = SMOTEENN(enn = EditedNearestNeighbours(sampling_strategy='all'), n_jobs =-1)\n        \n    def preprocess_train(self, X_train, y_train):\n        X_train = self.preprocess1.fit_transform(X_train)\n        X_train = pd.DataFrame(X_train, columns=self.preprocess1.get_feature_names_out())\n        X_train = self.preprocess2.fit_transform(X_train)\n        X_train = pd.DataFrame(X_train, columns=self.preprocess2.get_feature_names_out())\n        X_train, y_train = self.sm.fit_resample(X = X_train, y = y_train)\n        return X_train, y_train\n\n    def preprocess_test(self, X_test):\n        X_test = self.preprocess1.transform(X_test)\n        X_test = pd.DataFrame(X_test, columns=self.preprocess1.get_feature_names_out())\n        X_test = self.preprocess2.transform(X_test)\n        X_test = pd.DataFrame(X_test, columns=self.preprocess2.get_feature_names_out())\n        return X_test","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:37.809442Z","iopub.execute_input":"2022-07-29T04:51:37.810372Z","iopub.status.idle":"2022-07-29T04:51:37.825138Z","shell.execute_reply.started":"2022-07-29T04:51:37.810337Z","shell.execute_reply":"2022-07-29T04:51:37.824003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df.drop(\"income\", axis=1)\ny = (df[\"income\"] == \">50K\")\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)\nmy_preprocess = MyPreprocess()\nX_train, y_train = my_preprocess.preprocess_train(X_train, y_train)\nX_test = my_preprocess.preprocess_test(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:51:37.826490Z","iopub.execute_input":"2022-07-29T04:51:37.826850Z","iopub.status.idle":"2022-07-29T04:52:35.443575Z","shell.execute_reply.started":"2022-07-29T04:51:37.826816Z","shell.execute_reply":"2022-07-29T04:52:35.442059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"models = {\n    \"rfc\": RandomForestClassifier(),\n    \"abc\": AdaBoostClassifier(),\n    \"gbc\": GradientBoostingClassifier(),\n    \"etc\": ExtraTreesClassifier(),\n    \"dtc\": DecisionTreeClassifier(),\n    \"knc\": KNeighborsClassifier(),\n    \"svc\": SVC(),\n}","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:52:35.445093Z","iopub.execute_input":"2022-07-29T04:52:35.445432Z","iopub.status.idle":"2022-07-29T04:52:35.452003Z","shell.execute_reply.started":"2022-07-29T04:52:35.445401Z","shell.execute_reply":"2022-07-29T04:52:35.450617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cvs = dict()\n\nfor model_name, model in models.items():\n    print(model_name)\n    cv = cross_val_score(model, X_train, y_train, cv=StratifiedKFold(n_splits=5), scoring=\"roc_auc\")\n    print(\"Average roc_auc score:\", np.mean(cv))\n    print(\"Standard deviation:\", np.std(cv))\n    cvs[model_name] = cv","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:52:35.453833Z","iopub.execute_input":"2022-07-29T04:52:35.454197Z","iopub.status.idle":"2022-07-29T04:55:40.414624Z","shell.execute_reply.started":"2022-07-29T04:52:35.454165Z","shell.execute_reply":"2022-07-29T04:55:40.413354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Extra Trees Classifier seemed to give the highest roc_auc score. Hence a grid search is performed on its parameters.","metadata":{}},{"cell_type":"code","source":"def grid_search(model, params_dict, X_train, y_train, verbose=True):\n    \"\"\"\n    Perform grid search on a model by given model, parameters dict, training data (X and y).\n    \"\"\"\n    \n    grid = GridSearchCV(model, params_dict, scoring=\"roc_auc\", cv=StratifiedKFold(n_splits=5))\n    grid.fit(X_train, y_train)\n    \n    if verbose:\n        print(\"Best Params: \", grid.best_params_)\n        print(\"Best Recall: \", grid.best_score_)\n    \n    return grid","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:55:40.416010Z","iopub.execute_input":"2022-07-29T04:55:40.416839Z","iopub.status.idle":"2022-07-29T04:55:40.423816Z","shell.execute_reply.started":"2022-07-29T04:55:40.416804Z","shell.execute_reply":"2022-07-29T04:55:40.422669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"etc_dict = {\n    \"n_estimators\": [50, 100, 200],\n    \"criterion\": [\"gini\", \"entropy\"],\n    \"max_depth\": [10, 20, 40, None]\n}\n\ngrid = grid_search(models[\"etc\"], etc_dict, X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:55:40.425632Z","iopub.execute_input":"2022-07-29T04:55:40.426170Z","iopub.status.idle":"2022-07-29T05:03:03.084000Z","shell.execute_reply.started":"2022-07-29T04:55:40.426126Z","shell.execute_reply":"2022-07-29T05:03:03.082865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def report_model(model, X_test, y_test):\n    \"\"\"\n    Given model and validation data (X and y), report the performance of the model, including:\n    ROC Plot\n    Final ROC-AUC score\n    Final Confusion Matrix\n    \"\"\"\n    \n    y_predict = model.predict(X_test)\n    \n    fpr, tpr, thresholds = roc_curve(y_test, y_predict)\n    display = RocCurveDisplay(fpr=fpr, tpr=tpr)\n    display.plot()\n    plt.show()\n\n    print(\"Final Test ROC-AUC: \", roc_auc_score(y_test, y_predict))\n\n    cm = confusion_matrix(y_test, y_predict)\n\n    print(\"\\nFinal Confusion Matrix: \\n\")\n    print(cm)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T05:03:03.085384Z","iopub.execute_input":"2022-07-29T05:03:03.085727Z","iopub.status.idle":"2022-07-29T05:03:03.093421Z","shell.execute_reply.started":"2022-07-29T05:03:03.085695Z","shell.execute_reply":"2022-07-29T05:03:03.092302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_model = ExtraTreesClassifier(**grid.best_params_)\nfinal_model.fit(X_train, y_train)\n\nreport_model(final_model, X_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T05:03:03.094795Z","iopub.execute_input":"2022-07-29T05:03:03.095512Z","iopub.status.idle":"2022-07-29T05:03:12.893066Z","shell.execute_reply.started":"2022-07-29T05:03:03.095478Z","shell.execute_reply":"2022-07-29T05:03:12.892009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"markdown","source":"Now to prepare competition testing data. Remember that there were null values in the testing data. Medians for numerical columns and modes for categorical columns are used.","metadata":{}},{"cell_type":"code","source":"df_predict = df_test.drop(\"Unnamed: 0\", axis=1)\ndf_predict = df_predict.fillna(df.select_dtypes(include=\"number\").median())\ndf_predict = df_predict.fillna(df.select_dtypes(exclude=\"number\").mode().iloc[0])\n    \ndf_predict.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T05:03:12.894293Z","iopub.execute_input":"2022-07-29T05:03:12.894607Z","iopub.status.idle":"2022-07-29T05:03:13.073886Z","shell.execute_reply.started":"2022-07-29T05:03:12.894578Z","shell.execute_reply":"2022-07-29T05:03:13.073028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_predict = my_preprocess.preprocess_test(df_predict)\ny_predict = final_model.predict(X_predict)\ny_predict","metadata":{"execution":{"iopub.status.busy":"2022-07-29T05:03:13.075200Z","iopub.execute_input":"2022-07-29T05:03:13.075725Z","iopub.status.idle":"2022-07-29T05:03:15.476021Z","shell.execute_reply.started":"2022-07-29T05:03:13.075683Z","shell.execute_reply":"2022-07-29T05:03:15.474869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission = pd.read_csv(\"/kaggle/input/udacity-mlcharity-competition/example_submission.csv\")\ndf_submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T05:03:15.477496Z","iopub.execute_input":"2022-07-29T05:03:15.478078Z","iopub.status.idle":"2022-07-29T05:03:15.502110Z","shell.execute_reply.started":"2022-07-29T05:03:15.478042Z","shell.execute_reply":"2022-07-29T05:03:15.500978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission[\"income\"] = y_predict.astype(int)\ndf_submission = df_submission.set_index(\"id\")\ndf_submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T05:03:15.503753Z","iopub.execute_input":"2022-07-29T05:03:15.504469Z","iopub.status.idle":"2022-07-29T05:03:15.518044Z","shell.execute_reply.started":"2022-07-29T05:03:15.504423Z","shell.execute_reply":"2022-07-29T05:03:15.516980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T05:03:15.519517Z","iopub.execute_input":"2022-07-29T05:03:15.520157Z","iopub.status.idle":"2022-07-29T05:03:15.620128Z","shell.execute_reply.started":"2022-07-29T05:03:15.520122Z","shell.execute_reply":"2022-07-29T05:03:15.619061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Acknowledgement\nSpecial thanks to Jose Guzman's [work](https://www.kaggle.com/code/joseguzman/8-models-that-predict-autism-spectrum-disorder) on Autism Prediction competition\nAll the great notebooks shared on kaggles","metadata":{}}]}