{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook displays the evolvement of numeric features over time for both classes.\n\nReferences  \n-https://www.kaggle.com/code/pavelvod/amex-eda-even-more-insane-time-patterns-revealed  \n-https://www.kaggle.com/code/pavelvod/amex-eda-revealing-time-patterns-of-features/notebook\n","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport datetime\nimport warnings\n#from colorama import Fore, Back, Style\nfrom scipy.stats import gmean\nfrom datetime import datetime, timedelta\nimport seaborn as sns\nimport math\n","metadata":{"id":"_IS3UFflaHEu","execution":{"iopub.status.busy":"2022-06-21T16:33:34.531021Z","iopub.execute_input":"2022-06-21T16:33:34.531375Z","iopub.status.idle":"2022-06-21T16:33:34.537112Z","shell.execute_reply.started":"2022-06-21T16:33:34.531347Z","shell.execute_reply":"2022-06-21T16:33:34.535892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_parquet('../input/amex-data-integer-dtypes-parquet-format/train.parquet')\n#test = pd.read_feather('test_data.ftr')\n\ntarget = pd.read_csv(\"../input/amex-default-prediction/train_labels.csv\")\n\n\nbin_cols = ['B_31', 'D_87']\ncat_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\nnum_cols = list(set(train.columns)-set(cat_cols+['S_2', 'customer_ID']))\n\n\ntrain['S_2'] = pd.to_datetime(train['S_2'])\ntrain['S_2_max'] = train[['S_2','customer_ID']].groupby('customer_ID').S_2.transform('max')\ntrain['S_2_min'] = train[['S_2','customer_ID']].groupby('customer_ID').S_2.transform('min')\n#train['S_2_diff'] = train[['S_2','customer_ID']].groupby('customer_ID').S_2.transform('diff').dt.days\ntrain['S_2_d'] = (train['S_2_max']-train['S_2']).dt.days\n#train['S_2_m'] = round((train['S_2']-train['S_2_min'])/np.timedelta64(1, 'M'), 0)\ntrain['S_2_m'] = (train['S_2']-train['S_2_min'])\ntrain['S_2_m'] = train[['S_2_m','customer_ID']].groupby('customer_ID').rank()\n\ntrain = pd.merge(train, target, on='customer_ID', how ='left')\ntrain.sort_values(['customer_ID', 'S_2_d'],  ascending = [True, False], inplace = True)\n","metadata":{"id":"HDcevwMRaPDl","execution":{"iopub.status.busy":"2022-06-21T16:33:34.548715Z","iopub.execute_input":"2022-06-21T16:33:34.54946Z","iopub.status.idle":"2022-06-21T16:35:17.432932Z","shell.execute_reply.started":"2022-06-21T16:33:34.549421Z","shell.execute_reply":"2022-06-21T16:35:17.431831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will only check for those creditors that provided 13 statements, hence were present during the whole time period.","metadata":{}},{"cell_type":"code","source":"df_presence = train.groupby(['customer_ID']).size().reset_index().rename(columns={0:'presence'})\ndf_presence","metadata":{"id":"zTaPPQm_ykQT","outputId":"a859368a-bba9-453e-d67a-c5a7c8355496","execution":{"iopub.status.busy":"2022-06-21T16:35:17.437419Z","iopub.execute_input":"2022-06-21T16:35:17.437893Z","iopub.status.idle":"2022-06-21T16:35:18.677227Z","shell.execute_reply.started":"2022-06-21T16:35:17.437852Z","shell.execute_reply":"2022-06-21T16:35:18.67612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.merge(train, df_presence, on='customer_ID', how ='left')","metadata":{"id":"nyA8hGUWmiNZ","execution":{"iopub.status.busy":"2022-06-21T16:35:18.68623Z","iopub.execute_input":"2022-06-21T16:35:18.686924Z","iopub.status.idle":"2022-06-21T16:35:23.000191Z","shell.execute_reply.started":"2022-06-21T16:35:18.686874Z","shell.execute_reply":"2022-06-21T16:35:22.999076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Records are aggregated on month level and plotted seperately for defaulting and non-defaulting (performing) customers","metadata":{}},{"cell_type":"code","source":"\n\ni,j=0,0\nPLOTS_PER_ROW = 5\nfig, axs = plt.subplots(math.ceil(len(num_cols)/PLOTS_PER_ROW),PLOTS_PER_ROW, figsize=(20, 80))\nfor col in num_cols:\n  temp1 = train[(train['target']==0) & (train['presence']==13)].groupby('S_2_m')[col].mean().reset_index().set_index('S_2_m')\n  temp2 = train[(train['target']==1) & (train['presence']==13)].groupby('S_2_m')[col].mean().reset_index().set_index('S_2_m')\n  temp1.columns= ['Performing']\n  temp2.columns= ['Default']\n  sns.lineplot(data = pd.merge(temp1, temp2, on = 'S_2_m'), palette = ['green', 'red'], ax=axs[i][j])\n  axs[i][j].set(xlabel='Months',\n       ylabel=col)\n  j+=1\n  if j%PLOTS_PER_ROW==0:\n    i+=1\n    j=0\nplt.show()\n\n\n\n\n\n","metadata":{"id":"m_XBRuAvhrv1","outputId":"5de74c45-3242-4525-e20a-29c95c1ebdb5","execution":{"iopub.status.busy":"2022-06-21T16:35:23.001905Z","iopub.execute_input":"2022-06-21T16:35:23.002297Z","iopub.status.idle":"2022-06-21T16:41:17.329012Z","shell.execute_reply.started":"2022-06-21T16:35:23.002259Z","shell.execute_reply":"2022-06-21T16:41:17.327034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Insight\n- not only levels but also growth rates are very differnt for both classes, giving nice new features","metadata":{}}]}