{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport pickle, gc\nfrom matplotlib import pyplot as plt","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-11-29T18:00:03.791666Z","iopub.execute_input":"2022-11-29T18:00:03.792272Z","iopub.status.idle":"2022-11-29T18:00:03.799916Z","shell.execute_reply.started":"2022-11-29T18:00:03.792229Z","shell.execute_reply":"2022-11-29T18:00:03.798399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = pd.read_csv('../input/amex-default-prediction/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2022-11-29T18:00:03.801946Z","iopub.execute_input":"2022-11-29T18:00:03.802501Z","iopub.status.idle":"2022-11-29T18:00:04.547113Z","shell.execute_reply.started":"2022-11-29T18:00:03.802449Z","shell.execute_reply":"2022-11-29T18:00:04.54579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def0 = len(train_labels[train_labels[\"target\"]==0])\ndef1 = len(train_labels[train_labels[\"target\"]==1])\nperc = def0/(def0+def1)\nplt.pie([def0, def1], labels=['no default', 'default'])\nprint('Percentage of No-Defaults = {}'.format(perc))\n\nplt.title('Label Distribution in Training Set')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-29T18:00:04.548911Z","iopub.execute_input":"2022-11-29T18:00:04.549307Z","iopub.status.idle":"2022-11-29T18:00:04.695489Z","shell.execute_reply.started":"2022-11-29T18:00:04.549271Z","shell.execute_reply":"2022-11-29T18:00:04.69381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain = pd.read_feather('../input/amexfeather/train_data.ftr')\ntest = pd.read_feather('../input/amexfeather/test_data.ftr')\n\nwith pd.option_context(\"display.min_rows\", 3):\n    display(train)\n    display(test)","metadata":{"execution":{"iopub.status.busy":"2022-11-29T18:00:04.697924Z","iopub.execute_input":"2022-11-29T18:00:04.699656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Train statement dates: ', train.S_2.min(), train.S_2.max())\nprint('Test statement dates: ',  test.S_2.min(), test.S_2.max())\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info(max_cols=200, show_counts=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Counting the statements per customer","metadata":{}},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 5))\ntrain_sc = train.customer_ID.value_counts().value_counts().sort_index(ascending=False).rename('Train statements per customer')\nax1.pie(train_sc, labels=train_sc.index)\nax1.set_title(train_sc.name)\ntest_sc = test.customer_ID.value_counts().value_counts().sort_index(ascending=False).rename('Test statements per customer')\nax2.pie(test_sc, labels=test_sc.index)\nax2.set_title(test_sc.name)\nplt.show()\n\n# display(train.customer_ID.value_counts().value_counts().sort_index(ascending=False).rename('Train statements per customer'))\n# display(train.customer_ID.value_counts().value_counts().sort_index(ascending=False).rename('Test statements per customer'))\n","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = train.S_2.groupby(train.customer_ID).max()\nplt.figure(figsize=(16, 4))\nplt.hist(temp, bins=pd.date_range(\"2018-03-01\", \"2018-04-01\", freq=\"d\"),\n         rwidth=0.8)\nplt.title('Train Customers Last Statement Date', fontsize=20)\nplt.xlabel('Last statement date per customer')\nplt.ylabel('Count')\nplt.gca()\nplt.show()\ndel temp\n\ntemp = test.S_2.groupby(test.customer_ID).max()\nplt.figure(figsize=(16, 4))\nplt.hist(temp, bins=pd.date_range(\"2019-04-01\", \"2019-11-01\", freq=\"d\"),\n         rwidth=0.74)\nplt.title('Test Customers Last Statement Date', fontsize=20)\nplt.xlabel('Last statement date per customer')\nplt.ylabel('Count')\nplt.gca()\nplt.show()\ndel temp","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = train.S_2.groupby(train.customer_ID).agg(['max', 'min'])\nplt.figure(figsize=(16, 3))\nplt.hist((temp['max'] - temp['min']).dt.days, bins=400)\nplt.xlabel('days')\nplt.ylabel('count')\nplt.title('Number of days between first and last statement of customer (train)', fontsize=20)\nplt.gca()\nplt.show()\n\ntemp = test.S_2.groupby(test.customer_ID).agg(['max', 'min'])\nplt.figure(figsize=(16, 3))\nplt.hist((temp['max'] - temp['min']).dt.days, bins=400)\nplt.xlabel('days')\nplt.ylabel('count')\nplt.title('Number of days between first and last statement of customer (test)', fontsize=20)\nplt.gca()\nplt.show()\ndel temp","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Categorical features\n","metadata":{}},{"cell_type":"code","source":"cat_features = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\nplt.figure(figsize=(16, 16))\nfor i, f in enumerate(cat_features):\n    plt.subplot(4, 3, i+1)\n    temp = pd.DataFrame(train[f][train.target == 0].value_counts(dropna=False, normalize=True).sort_index().rename('count'))\n    temp.index.name = 'value'\n    temp.reset_index(inplace=True)\n    plt.bar(temp.index, temp['count'], alpha=0.5, label='target=0')\n    temp = pd.DataFrame(train[f][train.target == 1].value_counts(dropna=False, normalize=True).sort_index().rename('count'))\n    temp.index.name = 'value'\n    temp.reset_index(inplace=True)\n    plt.bar(temp.index, temp['count'], alpha=0.5, label='target=1')\n    plt.xlabel(f)\n    plt.ylabel('frequency')\n    plt.legend()\n    plt.xticks(temp.index, temp.value)\nplt.suptitle('Categorical features', fontsize=20, y=0.93)\nplt.show()\ndel temp\n","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Quantitative features\n","metadata":{}},{"cell_type":"code","source":"cont_features = sorted([f for f in train.columns if f not in cat_features + bin_features + ['customer_ID', 'target', 'S_2']])\nprint(len(cont_features))\n# print(cont_features)\nncols = 4\nfor i, f in enumerate(cont_features):\n    if i % ncols == 0: \n        if i > 0: plt.show()\n        plt.figure(figsize=(16, 3))\n        if i == 0: plt.suptitle('Continuous features', fontsize=20, y=1.02)\n    plt.subplot(1, ncols, i % ncols + 1)\n    plt.hist(train[f], bins=200)\n    plt.xlabel(f)\nplt.show()","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]}]}