{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-05T08:38:31.917132Z","iopub.execute_input":"2022-06-05T08:38:31.917492Z","iopub.status.idle":"2022-06-05T08:38:31.951617Z","shell.execute_reply.started":"2022-06-05T08:38:31.917418Z","shell.execute_reply":"2022-06-05T08:38:31.950921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%reload_ext autoreload\n%autoreload 2\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-06-05T08:38:33.14376Z","iopub.execute_input":"2022-06-05T08:38:33.144331Z","iopub.status.idle":"2022-06-05T08:38:33.200788Z","shell.execute_reply.started":"2022-06-05T08:38:33.144288Z","shell.execute_reply":"2022-06-05T08:38:33.199992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport sklearn\nfrom fastai.tabular import *\nfrom fastai import *\nfrom pandas.api.types import is_string_dtype, is_numeric_dtype, is_categorical_dtype","metadata":{"execution":{"iopub.status.busy":"2022-06-05T08:38:36.437708Z","iopub.execute_input":"2022-06-05T08:38:36.438076Z","iopub.status.idle":"2022-06-05T08:38:36.956541Z","shell.execute_reply.started":"2022-06-05T08:38:36.438046Z","shell.execute_reply":"2022-06-05T08:38:36.955683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\nfrom pandas.api.types import is_string_dtype, is_numeric_dtype, is_categorical_dtype\nfrom fastai.tabular.all import *\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.tree import DecisionTreeRegressor\n#from dtreeviz.trees import *\nfrom IPython.display import Image, display_svg, SVG\nfrom sklearn import preprocessing\nimport gc\nimport random\npd.options.display.max_rows = 20\npd.options.display.max_columns = 8","metadata":{"execution":{"iopub.status.busy":"2022-06-05T08:38:37.602448Z","iopub.execute_input":"2022-06-05T08:38:37.602792Z","iopub.status.idle":"2022-06-05T08:38:39.831281Z","shell.execute_reply.started":"2022-06-05T08:38:37.602756Z","shell.execute_reply":"2022-06-05T08:38:39.830364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cupy\nimport cudf","metadata":{"execution":{"iopub.status.busy":"2022-06-05T08:38:39.833086Z","iopub.execute_input":"2022-06-05T08:38:39.833719Z","iopub.status.idle":"2022-06-05T08:38:42.239777Z","shell.execute_reply.started":"2022-06-05T08:38:39.833676Z","shell.execute_reply":"2022-06-05T08:38:42.238981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NAN_VALUE = -99","metadata":{"execution":{"iopub.status.busy":"2022-06-05T08:38:42.241052Z","iopub.execute_input":"2022-06-05T08:38:42.241401Z","iopub.status.idle":"2022-06-05T08:38:42.287871Z","shell.execute_reply.started":"2022-06-05T08:38:42.241364Z","shell.execute_reply":"2022-06-05T08:38:42.287076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Making use of the GPU library. This only works for integer only features at present.\ndef read_file_int(path = '', usecols = None):\n    # LOAD DATAFRAME\n    if usecols is not None: df = cudf.read_feather(path, columns=usecols)\n    else: df = cudf.read_feather(path)\n    # REDUCE DTYPE FOR CUSTOMER AND DATE\n#   df['customer_ID'] = df['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    df.S_2 = cudf.to_datetime(df.S_2)\n    # CREATE OVERALL ROW MISS VALUE\n    features = [x for x in df.columns.values if x not in ['customer_ID', 'target']]\n    df['n_missing'] = df[features].isna().sum(axis=1)\n    # FILL NAN\n    #df = df.fillna(NAN_VALUE) \n    # KEEP ONLY FINAL CUSTOMER ID UNTIL FUTURE TIME SERIES WORK BEGINS\n    df_out = df.groupby(['customer_ID']).tail(1).reset_index(drop=True)\n    print('shape of data:', df_out.shape)\n    del df\n    return df_out\n\n# To ensure that the categorical features are imported only using CPU\ndef read_file_cpu(path = '', usecols = None):\n    # LOAD DATAFRAME\n    if usecols is not None: df = pd.read_feather(path, columns=usecols)\n    else: df = pd.read_feather(path)\n    # REDUCE DTYPE FOR CUSTOMER AND DATE\n#   df['customer_ID'] = df['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    df.S_2 = pd.to_datetime(df.S_2)\n    # CREATE OVERALL ROW MISS VALUE\n    features = [x for x in df.columns.values if x not in ['customer_ID', 'target']]\n    df['n_missing'] = df[features].isna().sum(axis=1)\n    # FILL NAN\n     \n    features_num = [x for x in df._get_numeric_data().columns.values if x not in ['customer_ID', 'target']]\n    df = df[features_num].fillna(NAN_VALUE) \n    # KEEP ONLY FINAL CUSTOMER ID UNTIL FUTURE TIME SERIES WORK BEGINS\n    df_out = df.groupby(['customer_ID']).tail(1).reset_index(drop=True)\n    print('shape of data:', df_out.shape)\n    del df\n    return df_out\n\nprint('Reading train data...')\nTRAIN_PATH = '../input/amexfeather/train_data.ftr'\ntrain_df = read_file_cpu(path = TRAIN_PATH)\n\nprint('Reading test data...')\nTEST_PATH = '../input/amexfeather/test_data.ftr'\ntest_df = read_file_cpu(path = TEST_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-06-05T08:40:32.854953Z","iopub.execute_input":"2022-06-05T08:40:32.855322Z","iopub.status.idle":"2022-06-05T08:40:58.386194Z","shell.execute_reply.started":"2022-06-05T08:40:32.855284Z","shell.execute_reply":"2022-06-05T08:40:58.384723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Understanding the file size of one file\nfrom humanize import naturalsize\nsize = test_df.memory_usage(deep='True').sum()\nprint(size)\nprint(naturalsize(size))","metadata":{"execution":{"iopub.status.busy":"2022-06-05T07:03:06.024742Z","iopub.execute_input":"2022-06-05T07:03:06.02548Z","iopub.status.idle":"2022-06-05T07:03:07.120382Z","shell.execute_reply.started":"2022-06-05T07:03:06.025439Z","shell.execute_reply":"2022-06-05T07:03:07.119552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Train data memory usage: {naturalsize(test_df.memory_usage(deep=\"True\").sum())} ')\nprint(f'Test data memory usage:  {naturalsize(test_df.memory_usage(deep=\"True\").sum())}')","metadata":{"execution":{"iopub.status.busy":"2022-06-05T07:03:07.121352Z","iopub.execute_input":"2022-06-05T07:03:07.121697Z","iopub.status.idle":"2022-06-05T07:03:07.656152Z","shell.execute_reply.started":"2022-06-05T07:03:07.121661Z","shell.execute_reply":"2022-06-05T07:03:07.655178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head","metadata":{"execution":{"iopub.status.busy":"2022-06-05T07:03:07.662289Z","iopub.execute_input":"2022-06-05T07:03:07.665872Z","iopub.status.idle":"2022-06-05T07:03:07.816463Z","shell.execute_reply.started":"2022-06-05T07:03:07.665833Z","shell.execute_reply":"2022-06-05T07:03:07.815663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Reading train data...')\nTRAIN_PATH = '../input/amexfeather/train_data.ftr'\ntrain_df = read_file_cpu(path = TRAIN_PATH)\nprint(f\"Train NA Values\\n{train_df.isnull().sum()}\")\n\nprint('Reading test data...')\nTEST_PATH = '../input/amexfeather/test_data.ftr'\ntest_df = read_file_cpu(path = TEST_PATH)\nprint(f\"Test NA Values\\n{test_df.isnull().sum()}\")","metadata":{"execution":{"iopub.status.busy":"2022-06-05T07:04:06.665338Z","iopub.execute_input":"2022-06-05T07:04:06.665838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.columns","metadata":{"execution":{"iopub.status.busy":"2022-06-05T04:04:54.935738Z","iopub.execute_input":"2022-06-05T04:04:54.936237Z","iopub.status.idle":"2022-06-05T04:04:55.067088Z","shell.execute_reply.started":"2022-06-05T04:04:54.936185Z","shell.execute_reply":"2022-06-05T04:04:55.066075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = add_datepart(train_df, 'S_2')","metadata":{"execution":{"iopub.status.busy":"2022-06-05T04:04:55.805183Z","iopub.execute_input":"2022-06-05T04:04:55.805654Z","iopub.status.idle":"2022-06-05T04:04:57.1178Z","shell.execute_reply.started":"2022-06-05T04:04:55.805616Z","shell.execute_reply":"2022-06-05T04:04:57.116838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"' '.join(o for o in df.columns if o.startswith('S_2'))","metadata":{"execution":{"iopub.status.busy":"2022-06-05T04:04:57.122701Z","iopub.execute_input":"2022-06-05T04:04:57.124956Z","iopub.status.idle":"2022-06-05T04:04:57.198137Z","shell.execute_reply.started":"2022-06-05T04:04:57.124914Z","shell.execute_reply":"2022-06-05T04:04:57.197211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_columns = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']","metadata":{"execution":{"iopub.status.busy":"2022-06-05T04:04:57.305127Z","iopub.execute_input":"2022-06-05T04:04:57.307418Z","iopub.status.idle":"2022-06-05T04:04:57.377325Z","shell.execute_reply.started":"2022-06-05T04:04:57.307377Z","shell.execute_reply":"2022-06-05T04:04:57.376373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = df.columns.tolist()\ncont_columns = [column for column in columns if column not in cat_columns]\ncont_columns.remove('target')","metadata":{"execution":{"iopub.status.busy":"2022-06-05T04:04:58.575188Z","iopub.execute_input":"2022-06-05T04:04:58.575766Z","iopub.status.idle":"2022-06-05T04:04:58.678579Z","shell.execute_reply.started":"2022-06-05T04:04:58.575723Z","shell.execute_reply":"2022-06-05T04:04:58.677481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"procs = [Categorify, FillMissing]","metadata":{"execution":{"iopub.status.busy":"2022-06-05T04:04:59.485418Z","iopub.execute_input":"2022-06-05T04:04:59.48588Z","iopub.status.idle":"2022-06-05T04:04:59.557159Z","shell.execute_reply.started":"2022-06-05T04:04:59.485839Z","shell.execute_reply":"2022-06-05T04:04:59.556199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cond = (df.S_2Year<2018) | (df.S_2Month<1)\ntrain_idx = np.where( cond)[0]\nvalid_idx = np.where(~cond)[0]\n\nsplits = (list(train_idx),list(valid_idx))","metadata":{"execution":{"iopub.status.busy":"2022-06-05T04:05:00.315476Z","iopub.execute_input":"2022-06-05T04:05:00.315909Z","iopub.status.idle":"2022-06-05T04:05:00.446747Z","shell.execute_reply.started":"2022-06-05T04:05:00.315871Z","shell.execute_reply":"2022-06-05T04:05:00.445879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dep_var = 'target'","metadata":{"execution":{"iopub.status.busy":"2022-06-05T04:05:01.255397Z","iopub.execute_input":"2022-06-05T04:05:01.255845Z","iopub.status.idle":"2022-06-05T04:05:01.327536Z","shell.execute_reply.started":"2022-06-05T04:05:01.255806Z","shell.execute_reply":"2022-06-05T04:05:01.32636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat = cat_columns,\ncont = cont_columns","metadata":{"execution":{"iopub.status.busy":"2022-06-05T04:05:01.930093Z","iopub.execute_input":"2022-06-05T04:05:01.932775Z","iopub.status.idle":"2022-06-05T04:05:02.001671Z","shell.execute_reply.started":"2022-06-05T04:05:01.932733Z","shell.execute_reply":"2022-06-05T04:05:02.000537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"to = TabularPandas(df, procs, cat, cont, y_names=dep_var, splits=splits,reduce_memory=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-05T04:05:02.575267Z","iopub.execute_input":"2022-06-05T04:05:02.575738Z","iopub.status.idle":"2022-06-05T04:05:10.787139Z","shell.execute_reply.started":"2022-06-05T04:05:02.575699Z","shell.execute_reply":"2022-06-05T04:05:10.783775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(to.train),len(to.valid)","metadata":{"execution":{"iopub.status.busy":"2022-06-05T04:05:10.789986Z","iopub.status.idle":"2022-06-05T04:05:10.792219Z","shell.execute_reply.started":"2022-06-05T04:05:10.791965Z","shell.execute_reply":"2022-06-05T04:05:10.791991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"to.show(10)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"to.items.head(3)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"to.classes['D_63']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"to.classes['D_64']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xs,y = to.train.xs,to.train.y\nvalid_xs,valid_y = to.valid.xs,to.valid.y","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classifier = DecisionTreeClassifier(max_leaf_nodes=4)\nclassifier.fit(xs, y);","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"draw_tree(m, xs, size=10, leaves_parallel=True, precision=2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}