{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-29T21:01:09.824999Z","iopub.execute_input":"2022-07-29T21:01:09.825497Z","iopub.status.idle":"2022-07-29T21:01:10.992522Z","shell.execute_reply.started":"2022-07-29T21:01:09.825460Z","shell.execute_reply":"2022-07-29T21:01:10.991234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pathlib\n\nINPUT_DIR = '../input/parquet-files-amexdefault-prediction/'\nOUTPUT_DIR = ''\n\ntrain = pd.read_feather(pathlib.Path(INPUT_DIR + 'train_data.ftr'))\ntest = pd.read_feather(pathlib.Path(INPUT_DIR + 'test_data.ftr'))","metadata":{"execution":{"iopub.status.busy":"2022-07-29T21:01:10.995184Z","iopub.execute_input":"2022-07-29T21:01:10.995568Z","iopub.status.idle":"2022-07-29T21:02:04.517578Z","shell.execute_reply.started":"2022-07-29T21:01:10.995532Z","shell.execute_reply":"2022-07-29T21:02:04.516272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT_DIR = '../input/amex-default-prediction/'\nOUTPUT_DIR = ''\n\ntrain_labels=pd.read_csv(\"/kaggle/input/amex-default-prediction/train_labels.csv\")\nsample_submission=pd.read_csv(\"/kaggle/input/amex-default-prediction/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T21:02:04.519355Z","iopub.execute_input":"2022-07-29T21:02:04.519719Z","iopub.status.idle":"2022-07-29T21:02:07.645146Z","shell.execute_reply.started":"2022-07-29T21:02:04.519683Z","shell.execute_reply":"2022-07-29T21:02:07.643478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# merge train\n\ntrain_m = train.merge(train_labels,on='customer_ID',how='left')\ntrain_m.to_feather(pathlib.Path(OUTPUT_DIR + 'train_m.ftr'))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-29T21:02:07.648651Z","iopub.execute_input":"2022-07-29T21:02:07.649149Z","iopub.status.idle":"2022-07-29T21:03:47.755656Z","shell.execute_reply.started":"2022-07-29T21:02:07.649110Z","shell.execute_reply":"2022-07-29T21:03:47.753907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#D&P variables\n\nDelinquency = [c for c in train_m.columns if c.startswith('D_')]\nPayment = [c for c in train_m.columns if c.startswith('P_')]\nSpend = [c for c in train_m.columns if c.startswith('S_')]\nBalance = [c for c in train_m.columns if c.startswith('B_')]\nRisk = [c for c in train_m.columns if c.startswith('R_')]\n\ndef pie_chart():\n    labels = ['Delinquency', 'Payment', 'Spend', 'Balance', 'Risk']\n    size = [len(Delinquency), len(Payment), len(Spend), len(Balance), len(Risk)]\n    \n    fig = plt.pie(size, labels=labels, autopct='%1.1f%%', shadow=True, startangle=90)\n    return fig\n\npie_chart()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T21:03:47.757843Z","iopub.execute_input":"2022-07-29T21:03:47.758244Z","iopub.status.idle":"2022-07-29T21:03:48.044943Z","shell.execute_reply.started":"2022-07-29T21:03:47.758209Z","shell.execute_reply":"2022-07-29T21:03:48.043165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# BR\nbr = train_m.filter(regex='customer_ID|B|R|target')\nbr.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-29T21:03:48.047530Z","iopub.execute_input":"2022-07-29T21:03:48.048602Z","iopub.status.idle":"2022-07-29T21:03:54.180804Z","shell.execute_reply.started":"2022-07-29T21:03:48.048537Z","shell.execute_reply":"2022-07-29T21:03:54.179339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check missing values\nbr.isnull().sum()*100/len(br)\n\n#delete columns with more than 80% na\nperc = 80.0\nmin_count = int(((100-perc)/100)*br.shape[0]+1)\nbr = br.dropna(axis=1,thresh=min_count)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T21:03:54.182822Z","iopub.execute_input":"2022-07-29T21:03:54.184355Z","iopub.status.idle":"2022-07-29T21:03:59.675558Z","shell.execute_reply.started":"2022-07-29T21:03:54.184300Z","shell.execute_reply":"2022-07-29T21:03:59.674152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# target distribution\ntarget = ['0:paid','1:default']\ntarget_count = br['target'].value_counts()\nfig1, ax1 = plt.subplots()\nax1.pie(target_count,labels=target, autopct='%1.1f%%',\n        shadow=True, startangle=90)\nax1.axis('equal')  \n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T21:03:59.677438Z","iopub.execute_input":"2022-07-29T21:03:59.677796Z","iopub.status.idle":"2022-07-29T21:03:59.867851Z","shell.execute_reply.started":"2022-07-29T21:03:59.677761Z","shell.execute_reply":"2022-07-29T21:03:59.865970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# categorical variables B30\n#plt.style.use('_mpl-gallery')\nbr_type = dict(br.dtypes)\n\n# plot:\nfig, ax = plt.subplots()\nax.hist(br['B_30'],bins=3, linewidth=1, edgecolor=\"white\")\nax.set(xlim=(0,3),xticks=np.arange(0, 3),ylim=(0,(len(br)+1)))\nplt.show()\n\nsns.countplot(x = 'B_30',hue = 'target',data = br)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T21:03:59.870138Z","iopub.execute_input":"2022-07-29T21:03:59.870984Z","iopub.status.idle":"2022-07-29T21:04:00.954114Z","shell.execute_reply.started":"2022-07-29T21:03:59.870923Z","shell.execute_reply":"2022-07-29T21:04:00.953136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# categorical variables B38\nfig, ax = plt.subplots()\nax.hist(br['B_38'], bins=8, linewidth=0.5, edgecolor=\"white\")\nax.set(xlim=(0,8),xticks=np.arange(0, 8),ylim=(0,(len(br)+1)))\nplt.show()\n\nsns.countplot(x = 'B_38',hue = 'target',data = br)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T21:04:00.957482Z","iopub.execute_input":"2022-07-29T21:04:00.958662Z","iopub.status.idle":"2022-07-29T21:04:02.151488Z","shell.execute_reply.started":"2022-07-29T21:04:00.958619Z","shell.execute_reply":"2022-07-29T21:04:02.150417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# numerical variables and target distribution\n# drop categorical variables\nnum_br_target = br.drop(['B_30','B_38'],axis = 1)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-29T21:04:02.152835Z","iopub.execute_input":"2022-07-29T21:04:02.153988Z","iopub.status.idle":"2022-07-29T21:04:03.306609Z","shell.execute_reply.started":"2022-07-29T21:04:02.153948Z","shell.execute_reply":"2022-07-29T21:04:03.305422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# corr_heatmap function \n\ndef corr_heatmap(corr, title, size):\n    plt.figure(figsize=size)\n    mask = np.triu(corr)\n    sns.heatmap(corr, mask=mask, cmap='coolwarm', center = 0, vmin=-1, vmax=1, square=True, linewidths=0.5)\n    plt.title(title)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T21:04:03.308563Z","iopub.execute_input":"2022-07-29T21:04:03.309136Z","iopub.status.idle":"2022-07-29T21:04:03.318030Z","shell.execute_reply.started":"2022-07-29T21:04:03.309057Z","shell.execute_reply":"2022-07-29T21:04:03.316554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#distribution of balance variables and target\nb_target = br.filter(regex='target|B')\nb_target.head()\nb_cor = b_target.corr()\n\ncorr_heatmap(b_cor,'Balance Correlation Map', (25,25))","metadata":{"execution":{"iopub.status.busy":"2022-07-29T21:04:03.320103Z","iopub.execute_input":"2022-07-29T21:04:03.320717Z","iopub.status.idle":"2022-07-29T21:04:24.314958Z","shell.execute_reply.started":"2022-07-29T21:04:03.320646Z","shell.execute_reply":"2022-07-29T21:04:24.313582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#distribution of risk variables and target\nr_target = br.filter(regex='target|R')\nr_cor = r_target.corr()\ncorr_heatmap(r_cor,'Risk Correlation Map', (25,25))","metadata":{"execution":{"iopub.status.busy":"2022-07-29T21:04:24.316489Z","iopub.execute_input":"2022-07-29T21:04:24.316895Z","iopub.status.idle":"2022-07-29T21:04:36.529695Z","shell.execute_reply.started":"2022-07-29T21:04:24.316858Z","shell.execute_reply":"2022-07-29T21:04:36.528155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# D\ntrain_D = train_m.filter(regex='customer_ID|D|target')\ntest_D = test.filter(regex='customer_ID|D|target')\ntrain_D.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T21:04:36.531753Z","iopub.execute_input":"2022-07-29T21:04:36.532306Z","iopub.status.idle":"2022-07-29T21:04:38.204844Z","shell.execute_reply.started":"2022-07-29T21:04:36.532252Z","shell.execute_reply":"2022-07-29T21:04:38.203469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Category type variables in Delinquency \ncategory_D = train_D.select_dtypes(include = 'category')\nprint(category_D)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T21:04:38.206760Z","iopub.execute_input":"2022-07-29T21:04:38.207294Z","iopub.status.idle":"2022-07-29T21:04:38.237092Z","shell.execute_reply.started":"2022-07-29T21:04:38.207242Z","shell.execute_reply":"2022-07-29T21:04:38.236130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Distribution plots of category type variable\ndef category_distribution(list):\n    for col in list:\n        fig, ax = plt.subplots(1,2,figsize=(15,6))\n        fig.suptitle(col)\n        sns.countplot(ax=ax[0], x=col, hue='target', order = train_D[col].value_counts().index, palette= 'Pastel1', data=train_D)\n        ax[0].set_title('Train Data')\n        sns.countplot(ax=ax[1], x=col, order = test_D[col].value_counts().index, palette= 'Pastel1', data=test_D)\n        ax[1].set_title('Test Data')\n    plt.tight_layout()\n    plt.show()\n    \ncategory_distribution(category_D.columns.values.tolist())\n\n#count missing value\nprint(train_D.isnull().sum())\nprint(train_P.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-29T21:09:33.350009Z","iopub.execute_input":"2022-07-29T21:09:33.350481Z","iopub.status.idle":"2022-07-29T21:09:34.192806Z","shell.execute_reply.started":"2022-07-29T21:09:33.350446Z","shell.execute_reply":"2022-07-29T21:09:34.190907Z"},"trusted":true},"execution_count":null,"outputs":[]}]}