{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"},{"sourceId":3739819,"sourceType":"datasetVersion","datasetId":2231132}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-20T14:02:43.630584Z","iopub.execute_input":"2024-10-20T14:02:43.631041Z","iopub.status.idle":"2024-10-20T14:02:43.649219Z","shell.execute_reply.started":"2024-10-20T14:02:43.630998Z","shell.execute_reply":"2024-10-20T14:02:43.647941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LOAD LIBRARIES\nimport pandas as pd, numpy as np\nimport matplotlib.pyplot as plt\nfrom matplotlib import gridspec\n\n# LOAD TRAIN DATA AND MERGE TARGETS ONTO FEATURES\ndf = pd.read_csv('../input/amex-default-prediction/train_data.csv', nrows=100_000)\ndf.S_2 = pd.to_datetime(df.S_2)\ndf2 = pd.read_csv('../input/amex-default-prediction/train_labels.csv')\ndf = df.merge(df2,on='customer_ID',how='left')","metadata":{"execution":{"iopub.status.busy":"2024-10-20T14:02:45.651136Z","iopub.execute_input":"2024-10-20T14:02:45.651535Z","iopub.status.idle":"2024-10-20T14:02:50.702819Z","shell.execute_reply.started":"2024-10-20T14:02:45.651497Z","shell.execute_reply":"2024-10-20T14:02:50.701541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-20T14:02:52.766947Z","iopub.execute_input":"2024-10-20T14:02:52.767362Z","iopub.status.idle":"2024-10-20T14:02:52.793973Z","shell.execute_reply.started":"2024-10-20T14:02:52.767323Z","shell.execute_reply":"2024-10-20T14:02:52.792796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_time_series(prefix='D', cols=None, display_ct=32):\n\n    if cols is None:\n        COLS = df.columns[2:-1]\n        COLS = np.sort( [int(x[2:]) for x in COLS if x[0]==prefix] )\n        print(COLS)\n        COLS = [f'{prefix}_{x}' for x in COLS]\n        print('#'*25)\n        print(f'Plotting all {len(COLS)} columns with prefix {prefix}')\n        print('#'*25)\n    else:\n        COLS = [f'{prefix}_{x}' for x in cols]\n        print('#'*25)\n        print(f'Plotting {len(COLS)} columns with prefix {prefix}')\n        print('#'*25)\n        \n    for c in COLS:\n        tmp = df[['customer_ID','S_2',c,'target']].copy()\n        tmp2 = tmp.groupby(['customer_ID','target'])[['S_2',c]].agg(list).reset_index()\n        tmp3 = tmp2.loc[tmp2.target==1]\n        tmp4 = tmp2.loc[tmp2.target==0]\n\n        spec = gridspec.GridSpec(ncols=2, nrows=1,\n                             width_ratios=[3, 1], wspace=0.1,\n                             hspace=0.5, height_ratios=[1])\n        fig = plt.figure(figsize=(20,10))\n        ax0 = fig.add_subplot(spec[0])\n\n        # PLOT 32 DEFAULT CUSTOMERS AND 32 NON-DEFAULT CUSTOMERS\n        t0 = []; t1 = []\n        for k in range(display_ct):\n            try:\n                # PLOT DEFAULTING CUSTOMERS\n                row = tmp3.iloc[k]\n                ax0.plot(row.S_2,row[c],'-o',color='blue')\n                t1 += row[c]\n                # PLOT NON-DEFAULT CUSTOMERS\n                row = tmp4.iloc[k]\n                ax0.plot(row.S_2,row[c],'-o',color='red')\n                t0 += row[c]\n            except:\n                pass\n        plt.title(f'Feature {c} (Key: BLUE=DEFAULT, orange=no default)',size=18)\n\n        # PLOT HISTOGRAMS\n        ax1 = fig.add_subplot(spec[1])\n        try:\n            # COMPUTE BINS\n            t = t0+t1; mn = np.nanmin(t); mx = np.nanmax(t)\n            if mx==mn:\n                mx += 0.01; mn -= 0.01\n            bins = np.arange(mn,mx+(mx-mn)/20,(mx-mn)/20 )\n            # PLOT HISTOGRAMS\n            if np.sum(np.isnan(t1))!=len(t1):\n                ax1.hist(t1,bins=bins,orientation=\"horizontal\",alpha = 0.8,color='blue')\n            if np.sum(np.isnan(t0))!=len(t0):\n                ax1.hist(t0,bins=bins,orientation=\"horizontal\",alpha = 0.8,color='red')\n        except:\n            pass\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-20T14:02:55.090409Z","iopub.execute_input":"2024-10-20T14:02:55.090824Z","iopub.status.idle":"2024-10-20T14:02:55.107878Z","shell.execute_reply.started":"2024-10-20T14:02:55.090783Z","shell.execute_reply":"2024-10-20T14:02:55.106583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\n\ndef random_col_selection(prefix, catCols, numberOfCol):\n    # numberOfCol - Setting upper bound for number of col selection\n    cols = np.sort( [x for x in df.columns[2:-1] if x[0]==prefix] )\n    randomSelectedCols = []\n    if len(cols) < numberOfCol:\n        print(\"Randomly selected column count has to be less than total number of cols\")\n        return\n    for col in cols:\n        if len(randomSelectedCols) >= numberOfCol:\n            break\n        if random.randint(0, 1) and col not in catCols:\n            randomSelectedCols.append(col[2:])\n    \n    return randomSelectedCols","metadata":{"execution":{"iopub.status.busy":"2024-10-20T14:02:59.032062Z","iopub.execute_input":"2024-10-20T14:02:59.032472Z","iopub.status.idle":"2024-10-20T14:02:59.039986Z","shell.execute_reply.started":"2024-10-20T14:02:59.032430Z","shell.execute_reply":"2024-10-20T14:02:59.038623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categoricalCols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']","metadata":{"execution":{"iopub.status.busy":"2024-10-20T14:03:01.816024Z","iopub.execute_input":"2024-10-20T14:03:01.816474Z","iopub.status.idle":"2024-10-20T14:03:01.821866Z","shell.execute_reply.started":"2024-10-20T14:03:01.816434Z","shell.execute_reply":"2024-10-20T14:03:01.820668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot Deliquency Variables\nplot_time_series('D',random_col_selection('D', categoricalCols, 15))","metadata":{"execution":{"iopub.status.busy":"2024-10-20T14:03:03.567460Z","iopub.execute_input":"2024-10-20T14:03:03.568620Z","iopub.status.idle":"2024-10-20T14:03:26.737962Z","shell.execute_reply.started":"2024-10-20T14:03:03.568557Z","shell.execute_reply":"2024-10-20T14:03:26.736813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot Spend Variables\nplot_time_series('S',random_col_selection('S', categoricalCols, 15))","metadata":{"execution":{"iopub.status.busy":"2024-10-20T14:03:55.019307Z","iopub.execute_input":"2024-10-20T14:03:55.020270Z","iopub.status.idle":"2024-10-20T14:04:14.200063Z","shell.execute_reply.started":"2024-10-20T14:03:55.020223Z","shell.execute_reply":"2024-10-20T14:04:14.198746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot P_* Variables\nplot_time_series('P',random_col_selection('P', categoricalCols, 2))","metadata":{"execution":{"iopub.status.busy":"2024-10-20T14:04:22.575606Z","iopub.execute_input":"2024-10-20T14:04:22.576806Z","iopub.status.idle":"2024-10-20T14:04:25.917443Z","shell.execute_reply.started":"2024-10-20T14:04:22.576749Z","shell.execute_reply":"2024-10-20T14:04:25.916362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot B_* Variables\nplot_time_series('B',random_col_selection('B', categoricalCols, 10))","metadata":{"execution":{"iopub.status.busy":"2024-10-20T14:04:28.971546Z","iopub.execute_input":"2024-10-20T14:04:28.971978Z","iopub.status.idle":"2024-10-20T14:04:45.041674Z","shell.execute_reply.started":"2024-10-20T14:04:28.971935Z","shell.execute_reply":"2024-10-20T14:04:45.040466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot R_* Variables\nplot_time_series('R',random_col_selection('R', categoricalCols, 10))","metadata":{"execution":{"iopub.status.busy":"2024-10-20T14:04:49.319414Z","iopub.execute_input":"2024-10-20T14:04:49.319819Z","iopub.status.idle":"2024-10-20T14:05:05.356925Z","shell.execute_reply.started":"2024-10-20T14:04:49.319781Z","shell.execute_reply":"2024-10-20T14:05:05.355840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### In above plots we had visualized timeseries data of all types of variables. But since this is not the whole visualization of our data points. Since we only considered first 100_000 data. And we are only visualizaing timeseries data of 32 default customers and 32 non default customers. But this is a good start to analyze the data. Since our datasets was too large to analyze all the points at the one time. In deliquency variables you can see that there are some features which contain NULL Values and we are not able to plot visulatization for that particular feature.","metadata":{}},{"cell_type":"markdown","source":"#### So we have to remove NULL values and also we want to handle categorical values as well. I think here scaling is not needed since data is anoynmized and all the features data lies in between 0 to 1 with pretty much good precision points.","metadata":{}},{"cell_type":"markdown","source":"#### What's the next step?\n*  Find a way to incorporate this large amounts of data. In order to address this issue we have to either compress data file size by reducing feature number or feature data types. Or we can divide this large files into multiple files. And the performing each and every step on each and every file.","metadata":{}}]}