{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"based on https://www.kaggle.com/code/cdeotte/time-series-eda","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-19T23:59:00.574958Z","iopub.execute_input":"2022-08-19T23:59:00.575346Z","iopub.status.idle":"2022-08-19T23:59:00.605324Z","shell.execute_reply.started":"2022-08-19T23:59:00.575269Z","shell.execute_reply":"2022-08-19T23:59:00.604493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LOAD LIBRARIES\nimport pandas as pd, numpy as np\nimport matplotlib.pyplot as plt\nfrom matplotlib import gridspec\n\n# LOAD TRAIN DATA AND MERGE TARGETS ONTO FEATURES\ndf = pd.read_csv('../input/amex-default-prediction/train_data.csv', nrows=100_000)\ndf.S_2 = pd.to_datetime(df.S_2)\ndf2 = pd.read_csv('../input/amex-default-prediction/train_labels.csv')\ndf = df.merge(df2,on='customer_ID',how='left')","metadata":{"execution":{"iopub.status.busy":"2022-08-19T23:59:11.065585Z","iopub.execute_input":"2022-08-19T23:59:11.066353Z","iopub.status.idle":"2022-08-19T23:59:17.912030Z","shell.execute_reply.started":"2022-08-19T23:59:11.066321Z","shell.execute_reply":"2022-08-19T23:59:17.911025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_time_series(prefix='D', cols=None, display_ct=32):\n    \n    # DETERMINE WHICH COLUMNS TO PLOT\n    if cols is not None and len(cols)==0: cols = None\n    if cols is None:\n        COLS = df.columns[2:-1]\n        COLS = np.sort( [int(x[2:]) for x in COLS if x[0]==prefix] )\n        COLS = [f'{prefix}_{x}' for x in COLS]\n        print('#'*25)\n        print(f'Plotting all {len(COLS)} columns with prefix {prefix}')\n        print('#'*25)\n    else:\n        COLS = [f'{prefix}_{x}' for x in cols]\n        print('#'*25)\n        print(f'Plotting {len(COLS)} columns with prefix {prefix}')\n        print('#'*25)\n\n    # ITERATE COLUMNS\n    for c in COLS:\n\n        # CONVERT DATAFRAME INTO SERIES WITH COLUMN\n        tmp = df[['customer_ID','S_2',c,'target']].copy()\n        tmp2 = tmp.groupby(['customer_ID','target'])[['S_2',c]].agg(list).reset_index()\n        tmp3 = tmp2.loc[tmp2.target==1]\n        tmp4 = tmp2.loc[tmp2.target==0]\n\n        # FORMAT PLOT\n        spec = gridspec.GridSpec(ncols=2, nrows=1,\n                             width_ratios=[3, 1], wspace=0.1,\n                             hspace=0.5, height_ratios=[1])\n        fig = plt.figure(figsize=(20,10))\n        ax0 = fig.add_subplot(spec[0])\n\n        # PLOT 32 DEFAULT CUSTOMERS AND 32 NON-DEFAULT CUSTOMERS\n        t0 = []; t1 = []\n        for k in range(display_ct):\n            try:\n                # PLOT DEFAULTING CUSTOMERS\n                row = tmp3.iloc[k]\n                ax0.plot(row.S_2,row[c],'-o',color='blue')\n                #on my pc, I need to convert Timestamp to datetime: https://stackoverflow.com/questions/49947615/timestamp-overlapping-matplotlib\n                if False:\n                    ax0.plot([x.to_pydatetime() for x in row.S_2],row[c],'-o',color='blue')\n                t1 += row[c]\n                # PLOT NON-DEFAULT CUSTOMERS\n                row = tmp4.iloc[k]\n                ax0.plot(row.S_2,row[c],'-o',color='orange')\n                t0 += row[c]\n            except:\n                pass\n        plt.title(f'Feature {c} (Key: BLUE=DEFAULT, orange=no default)',size=18)\n\n        # PLOT HISTOGRAMS\n        ax1 = fig.add_subplot(spec[1])\n        try:\n            # COMPUTE BINS\n            t = t0+t1; mn = np.nanmin(t); mx = np.nanmax(t)\n            if mx==mn:\n                mx += 0.01; mn -= 0.01\n            bins = np.arange(mn,mx+(mx-mn)/20,(mx-mn)/20 )\n            # PLOT HISTOGRAMS\n            if np.sum(np.isnan(t1))!=len(t1):\n                ax1.hist(t1,bins=bins,orientation=\"horizontal\",alpha = 0.8,color='blue')\n            if np.sum(np.isnan(t0))!=len(t0):\n                ax1.hist(t0,bins=bins,orientation=\"horizontal\",alpha = 0.8,color='orange')\n        except:\n            pass\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-20T00:00:20.314823Z","iopub.execute_input":"2022-08-20T00:00:20.315138Z","iopub.status.idle":"2022-08-20T00:00:20.331797Z","shell.execute_reply.started":"2022-08-20T00:00:20.315115Z","shell.execute_reply":"2022-08-20T00:00:20.330640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LEAVE LIST BLANK TO PLOT ALL\nplot_time_series('D',[39,41,47,45,46,48,54,59,61,62,75,96,105,112,124])","metadata":{"execution":{"iopub.status.busy":"2022-08-20T00:00:29.020668Z","iopub.execute_input":"2022-08-20T00:00:29.021069Z","iopub.status.idle":"2022-08-20T00:00:46.282012Z","shell.execute_reply.started":"2022-08-20T00:00:29.021044Z","shell.execute_reply":"2022-08-20T00:00:46.280671Z"},"trusted":true},"execution_count":null,"outputs":[]}]}