{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\ntrain = pd.read_feather('../input/amexfeather/train_data.ftr')","metadata":{"execution":{"iopub.status.busy":"2022-07-19T09:16:15.313339Z","iopub.execute_input":"2022-07-19T09:16:15.314302Z","iopub.status.idle":"2022-07-19T09:16:33.205873Z","shell.execute_reply.started":"2022-07-19T09:16:15.314260Z","shell.execute_reply":"2022-07-19T09:16:33.204529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-19T09:16:33.208604Z","iopub.execute_input":"2022-07-19T09:16:33.209216Z","iopub.status.idle":"2022-07-19T09:16:33.220232Z","shell.execute_reply.started":"2022-07-19T09:16:33.209163Z","shell.execute_reply":"2022-07-19T09:16:33.218778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Following variables are categorical:\n'B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68'\n\nLet us explore each one of them","metadata":{}},{"cell_type":"code","source":"#check number of nulls \nprint('Number of nulls:',train['B_30'].isnull().sum())\nprint('Number of nulls of percentage of total :',train['B_30'].isnull().sum()/len(train)*100)\n#value counts\nprint(\"Value Counts:\")\ntrain['B_30'].value_counts().plot(kind='bar')\nplt.show()\nprint(\"Value Counts as percentage of total:\")\n(train['B_30'].value_counts()/len(train)*100).plot(kind='bar')\nplt.show()\nprint(\"Count of target when B_30 is 0:\\n\",train[train['B_30']==0]['target'].value_counts()/len(train[train['B_30']==0])*100)\n(train[train['B_30']==0]['target'].value_counts()/len(train[train['B_30']==0])*100).plot(kind='bar')\nplt.show()\nprint(\"Count of target when B_30 is 1:\\n\",train[train['B_30']==1]['target'].value_counts()/len(train[train['B_30']==1])*100)\n(train[train['B_30']==1]['target'].value_counts()/len(train[train['B_30']==1])*100).plot(kind='bar')\nplt.show()\nprint(\"Count of target when B_30 is 2:\\n\",train[train['B_30']==2]['target'].value_counts()/len(train[train['B_30']==2])*100)\n(train[train['B_30']==2]['target'].value_counts()/len(train[train['B_30']==2])*100).plot(kind='bar')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T09:16:33.221731Z","iopub.execute_input":"2022-07-19T09:16:33.223139Z","iopub.status.idle":"2022-07-19T09:17:16.646000Z","shell.execute_reply.started":"2022-07-19T09:16:33.223071Z","shell.execute_reply":"2022-07-19T09:17:16.644715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can see from the above analysis, in 85% of cases the value of B_30 is 0.\n\nWhen the value of B_30 is 0, in close to 82% of cases , the customer has not defulted. \n\nWhen the value of B_30 id 1, in close to 63% of cases the customer defaults. \n\nSimilarly , when the value of B_30 is 2, in close to 61% of cases the customer defaults.\n\nB_30 looks like a very useful variable. We can conclude that there is a more than 60% chance that the customer defaults if B_30 is greater than 0.","metadata":{}},{"cell_type":"code","source":"#check number of nulls \nprint('Number of nulls:',train['B_38'].isnull().sum())\nprint('Number of nulls of percentage of total :',train['B_38'].isnull().sum()/len(train)*100)\n#value counts\nprint(\"Value Counts:\")\nprint(train['B_38'].value_counts())\nprint(\"Value Counts as percentage of total:\")\nprint(train['B_38'].value_counts()/len(train)*100)\nlst_distinct=list(train[~train['B_38'].isnull()]['B_38'].unique())\nlst_distinct.sort()\nfor i in lst_distinct:\n    try:\n        print(\"Count of target when B_38 is \"+str(i)+\":\\n\",train[train['B_38']==i]['target'].value_counts()/len(train[train['B_38']==i])*100)\n    except:\n        pass\n#print(\"Count of target when B_30 is 1:\\n\",train[train['B_30']==1]['target'].value_counts()/len(train[train['B_30']==1])*100)\n#print(\"Count of target when B_30 is 2:\\n\",train[train['B_30']==2]['target'].value_counts()/len(train[train['B_30']==2])*100)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T09:17:16.648340Z","iopub.execute_input":"2022-07-19T09:17:16.649152Z","iopub.status.idle":"2022-07-19T09:17:49.618919Z","shell.execute_reply.started":"2022-07-19T09:17:16.649102Z","shell.execute_reply":"2022-07-19T09:17:49.617667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As is evident from the above data, the number of null values in B_38 is exactly same as that in B_30. However, the number of distinct values this variable takes in more than B_30.\n\nClose to 80% of values are fall in the range 1 to 3. Also, the number of non-defaults is markedly higher in this range. As we go beyond 3, the defaults percentage increases.\n\nCompared to B_30, B_38 is not so informative.","metadata":{}},{"cell_type":"code","source":"#check number of nulls \nprint('Number of nulls:',train['D_114'].isnull().sum())\nprint('Number of nulls of percentage of total :',train['D_114'].isnull().sum()/len(train)*100)\n#value counts\nprint(\"Value Counts:\")\nprint(train['D_114'].value_counts())\nprint(\"Value Counts as percentage of total:\")\nprint(train['D_114'].value_counts()/len(train)*100)\nlst_distinct=list(train[~train['D_114'].isnull()]['D_114'].unique())\nlst_distinct.sort()\nfor i in lst_distinct:\n    try:\n        print(\"Count of target when D_114 is \"+str(i)+\":\\n\",train[train['D_114']==i]['target'].value_counts()/len(train[train['D_114']==i])*100)\n    except:\n        pass\n#print(\"Count of target when B_30 is 1:\\n\",train[train['B_30']==1]['target'].value_counts()/len(train[train['B_30']==1])*100)\n#print(\"Count of target when B_30 is 2:\\n\",train[train['B_30']==2]['target'].value_counts()/len(train[train['B_30']==2])*100)\ntrain[train['target']==1]['D_114'].value_counts()/len(train[train['target']==1])*100","metadata":{"execution":{"iopub.status.busy":"2022-07-19T09:17:49.620331Z","iopub.execute_input":"2022-07-19T09:17:49.620860Z","iopub.status.idle":"2022-07-19T09:18:24.442202Z","shell.execute_reply.started":"2022-07-19T09:17:49.620822Z","shell.execute_reply":"2022-07-19T09:18:24.440837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Compared to B_30 and B_38 , number of null values is higher in D_114. \nWe see that when D_114 is 1, 82% of cases are non defaults. But given, that in general most of the cases are non-defaults, let us see what is distribution when default is 1.\n","metadata":{}},{"cell_type":"code","source":"#check number of nulls \nprint('Number of nulls:',train['D_116'].isnull().sum())\nprint('Number of nulls of percentage of total :',train['D_116'].isnull().sum()/len(train)*100)\n#value counts\nprint(\"Value Counts:\")\nprint(train['D_116'].value_counts())\nprint(\"Value Counts as percentage of total:\")\nprint(train['D_116'].value_counts()/len(train)*100)\nlst_distinct=list(train[~train['D_116'].isnull()]['D_116'].unique())\nlst_distinct.sort()\nfor i in lst_distinct:\n    try:\n        print(\"Count of target when D_116 is \"+str(i)+\":\\n\",train[train['D_116']==i]['target'].value_counts()/len(train[train['D_116']==i])*100)\n    except:\n        pass\n#print(\"Count of target when B_30 is 1:\\n\",train[train['B_30']==1]['target'].value_counts()/len(train[train['B_30']==1])*100)\n#print(\"Count of target when B_30 is 2:\\n\",train[train['B_30']==2]['target'].value_counts()/len(train[train['B_30']==2])*100)\ntrain[train['target']==1]['D_116'].value_counts()/len(train[train['target']==1])*100","metadata":{"execution":{"iopub.status.busy":"2022-07-19T09:20:10.341942Z","iopub.execute_input":"2022-07-19T09:20:10.343741Z","iopub.status.idle":"2022-07-19T09:20:51.101761Z","shell.execute_reply.started":"2022-07-19T09:20:10.343662Z","shell.execute_reply":"2022-07-19T09:20:51.100523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the above data, we can infer that when D_166 is 1 there is a higher probability of the card holder defaulting.","metadata":{}},{"cell_type":"code","source":"#check number of nulls \nprint('Number of nulls:',train['D_117'].isnull().sum())\nprint('Number of nulls of percentage of total :',train['D_117'].isnull().sum()/len(train)*100)\n#value counts\nprint(\"Value Counts:\")\nprint(train['D_117'].value_counts())\nprint(\"Value Counts as percentage of total:\")\nprint(train['D_117'].value_counts()/len(train)*100)\nlst_distinct=list(train[~train['D_117'].isnull()]['D_117'].unique())\nlst_distinct.sort()\nfor i in lst_distinct:\n    try:\n        print(\"Count of target when D_117 is \"+str(i)+\":\\n\",train[train['D_117']==i]['target'].value_counts()/len(train[train['D_117']==i])*100)\n    except:\n        pass\n#print(\"Count of target when B_30 is 1:\\n\",train[train['B_30']==1]['target'].value_counts()/len(train[train['B_30']==1])*100)\n#print(\"Count of target when B_30 is 2:\\n\",train[train['B_30']==2]['target'].value_counts()/len(train[train['B_30']==2])*100)\ntrain[train['target']==1]['D_117'].value_counts()/len(train[train['target']==1])*100","metadata":{"execution":{"iopub.status.busy":"2022-07-19T09:28:14.762026Z","iopub.execute_input":"2022-07-19T09:28:14.763912Z","iopub.status.idle":"2022-07-19T09:28:53.056597Z","shell.execute_reply.started":"2022-07-19T09:28:14.763816Z","shell.execute_reply":"2022-07-19T09:28:53.055066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This paramter too does not divulge much","metadata":{}},{"cell_type":"code","source":"#check number of nulls \nprint('Number of nulls:',train['D_120'].isnull().sum())\nprint('Number of nulls of percentage of total :',train['D_120'].isnull().sum()/len(train)*100)\n#value counts\nprint(\"Value Counts:\")\nprint(train['D_120'].value_counts())\nprint(\"Value Counts as percentage of total:\")\nprint(train['D_120'].value_counts()/len(train)*100)\nlst_distinct=list(train[~train['D_120'].isnull()]['D_120'].unique())\nlst_distinct.sort()\nfor i in lst_distinct:\n    try:\n        print(\"Count of target when D_120 is \"+str(i)+\":\\n\",train[train['D_120']==i]['target'].value_counts()/len(train[train['D_120']==i])*100)\n    except:\n        pass\n#print(\"Count of target when B_30 is 1:\\n\",train[train['B_30']==1]['target'].value_counts()/len(train[train['B_30']==1])*100)\n#print(\"Count of target when B_30 is 2:\\n\",train[train['B_30']==2]['target'].value_counts()/len(train[train['B_30']==2])*100)\ntrain[train['target']==1]['D_120'].value_counts()/len(train[train['target']==1])*100","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-19T09:34:56.055530Z","iopub.execute_input":"2022-07-19T09:34:56.056062Z","iopub.status.idle":"2022-07-19T09:35:37.497616Z","shell.execute_reply.started":"2022-07-19T09:34:56.056024Z","shell.execute_reply":"2022-07-19T09:35:37.496371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[train['D_114'].isnull()]['target'].value_counts()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-19T09:53:21.235507Z","iopub.execute_input":"2022-07-19T09:53:21.236026Z","iopub.status.idle":"2022-07-19T09:53:21.841077Z","shell.execute_reply.started":"2022-07-19T09:53:21.235986Z","shell.execute_reply":"2022-07-19T09:53:21.839529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We cabn safely create a function to repeat these stes for all delinquent variables that are categorical in nature.","metadata":{}},{"cell_type":"code","source":"lst_deli_var_cat=['D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\ndef analyse_deli_var_cat(i):\n    #check number of nulls \n    print('Number of nulls:',train[i].isnull().sum())\n    print('Number of nulls of percentage of total :',train[i].isnull().sum()/len(train)*100)\n#value counts\n    print(\"Value Counts:\")\n    print(train[i].value_counts())\n    print(\"Value Counts as percentage of total:\")\n    print(train[i].value_counts()/len(train)*100)\n    lst_distinct=list(train[~train[i].isnull()][i].unique())\n    lst_distinct.sort()\n    for j in lst_distinct:\n        try:\n            print(\"Count of target when \"+i+\" is \"+str(j)+\":\\n\",train[train[i]==j]['target'].value_counts()/len(train[train[i]==j])*100)\n        except:\n            pass\n#print(\"Count of target when B_30 is 1:\\n\",train[train['B_30']==1]['target'].value_counts()/len(train[train['B_30']==1])*100)\n#print(\"Count of target when B_30 is 2:\\n\",train[train['B_30']==2]['target'].value_counts()/len(train[train['B_30']==2])*100)\n    print(train[train['target']==1][i].value_counts()/len(train[train['target']==1])*100)\nfor i in lst_deli_var_cat:\n    analyse_deli_var_cat(i)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-19T10:27:38.275478Z","iopub.execute_input":"2022-07-19T10:27:38.276739Z","iopub.status.idle":"2022-07-19T10:32:53.407946Z","shell.execute_reply.started":"2022-07-19T10:27:38.276677Z","shell.execute_reply":"2022-07-19T10:32:53.406565Z"},"trusted":true},"execution_count":null,"outputs":[]}]}