{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport sklearn as skl\nimport random","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-29T09:03:18.555795Z","iopub.execute_input":"2022-05-29T09:03:18.556244Z","iopub.status.idle":"2022-05-29T09:03:19.642973Z","shell.execute_reply.started":"2022-05-29T09:03:18.556207Z","shell.execute_reply":"2022-05-29T09:03:19.641629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install fastai --upgrade","metadata":{"execution":{"iopub.status.busy":"2022-05-29T09:03:19.645046Z","iopub.execute_input":"2022-05-29T09:03:19.645547Z","iopub.status.idle":"2022-05-29T09:03:32.300389Z","shell.execute_reply.started":"2022-05-29T09:03:19.645498Z","shell.execute_reply":"2022-05-29T09:03:32.299145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from fastai.tabular.all import *","metadata":{"execution":{"iopub.status.busy":"2022-05-29T09:03:32.302815Z","iopub.execute_input":"2022-05-29T09:03:32.303346Z","iopub.status.idle":"2022-05-29T09:03:33.432343Z","shell.execute_reply.started":"2022-05-29T09:03:32.303290Z","shell.execute_reply":"2022-05-29T09:03:33.430239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# To avoid loading a 16GB datafile I randomly sample 50.000 rows from the train data\n# Change s if you want more rows :)","metadata":{"execution":{"iopub.status.busy":"2022-05-29T09:03:33.437721Z","iopub.execute_input":"2022-05-29T09:03:33.438307Z","iopub.status.idle":"2022-05-29T09:03:33.443974Z","shell.execute_reply.started":"2022-05-29T09:03:33.438270Z","shell.execute_reply":"2022-05-29T09:03:33.442445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random.seed(1) #fix random seed for consistency","metadata":{"execution":{"iopub.status.busy":"2022-05-29T09:03:33.445584Z","iopub.execute_input":"2022-05-29T09:03:33.446983Z","iopub.status.idle":"2022-05-29T09:03:33.457410Z","shell.execute_reply.started":"2022-05-29T09:03:33.446906Z","shell.execute_reply":"2022-05-29T09:03:33.455893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfilename = \"../input/amex-default-prediction/train_data.csv\"\nn = 5531451 #sum(1 for line in open(filename)) - 1 --> code to get n the first time. Number of records in file (excludes header)\ns = 50000 #desired sample size\nskip = sorted(random.sample(range(1,n+1),n-s)) #the 0-indexed header will not be included in the skip list\ndf = pd.read_csv(filename, skiprows=skip)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-29T09:03:33.459195Z","iopub.execute_input":"2022-05-29T09:03:33.459735Z","iopub.status.idle":"2022-05-29T09:07:11.143821Z","shell.execute_reply.started":"2022-05-29T09:03:33.459655Z","shell.execute_reply":"2022-05-29T09:07:11.142692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filename = \"../input/amex-default-prediction/train_labels.csv\"\ndf_y = pd.read_csv(filename, skiprows=skip)\ndf_y.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-29T09:07:11.145160Z","iopub.execute_input":"2022-05-29T09:07:11.145471Z","iopub.status.idle":"2022-05-29T09:07:12.134368Z","shell.execute_reply.started":"2022-05-29T09:07:11.145443Z","shell.execute_reply":"2022-05-29T09:07:12.132945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = df.join(df_y.set_index('customer_ID'),on='customer_ID')","metadata":{"execution":{"iopub.status.busy":"2022-05-29T10:59:09.925130Z","iopub.execute_input":"2022-05-29T10:59:09.926129Z","iopub.status.idle":"2022-05-29T10:59:10.005656Z","shell.execute_reply.started":"2022-05-29T10:59:09.926058Z","shell.execute_reply":"2022-05-29T10:59:10.003948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-29T10:59:17.213359Z","iopub.execute_input":"2022-05-29T10:59:17.213818Z","iopub.status.idle":"2022-05-29T10:59:17.328235Z","shell.execute_reply.started":"2022-05-29T10:59:17.213779Z","shell.execute_reply":"2022-05-29T10:59:17.327171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info(verbose=True, show_counts =True)","metadata":{"execution":{"iopub.status.busy":"2022-05-29T10:59:28.963124Z","iopub.execute_input":"2022-05-29T10:59:28.963545Z","iopub.status.idle":"2022-05-29T10:59:29.030979Z","shell.execute_reply.started":"2022-05-29T10:59:28.963512Z","shell.execute_reply":"2022-05-29T10:59:29.029790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = add_datepart(train_df, 'S_2')","metadata":{"execution":{"iopub.status.busy":"2022-05-29T10:59:35.829230Z","iopub.execute_input":"2022-05-29T10:59:35.830269Z","iopub.status.idle":"2022-05-29T10:59:36.055567Z","shell.execute_reply.started":"2022-05-29T10:59:35.830224Z","shell.execute_reply":"2022-05-29T10:59:36.054524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.to_csv('sample_train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-05-29T10:59:36.509536Z","iopub.execute_input":"2022-05-29T10:59:36.510184Z","iopub.status.idle":"2022-05-29T10:59:49.368727Z","shell.execute_reply.started":"2022-05-29T10:59:36.510134Z","shell.execute_reply":"2022-05-29T10:59:49.367581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_columns = ['B_30',\n 'B_38',\n 'D_114',\n 'D_116',\n 'D_117',\n 'D_120',\n 'D_126',\n 'D_66',\n'customer_ID',\n'D_64',\n'D_63',\n 'D_68', 'S_2Day', 'S_2Dayofweek', 'S_2Dayofyear', 'S_2Elapsed',\n       'S_2Is_month_end', 'S_2Is_month_start', 'S_2Is_quarter_end',\n       'S_2Is_quarter_start', 'S_2Is_year_end', 'S_2Is_year_start',\n       'S_2Month', 'S_2Week', 'S_2Year']","metadata":{"execution":{"iopub.status.busy":"2022-05-29T10:59:49.373272Z","iopub.execute_input":"2022-05-29T10:59:49.374206Z","iopub.status.idle":"2022-05-29T10:59:49.383235Z","shell.execute_reply.started":"2022-05-29T10:59:49.374134Z","shell.execute_reply":"2022-05-29T10:59:49.382077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = train_df.columns.tolist()\ncont_columns = [column for column in columns if column not in cat_columns]\ncont_columns.remove('target')","metadata":{"execution":{"iopub.status.busy":"2022-05-29T10:59:49.384733Z","iopub.execute_input":"2022-05-29T10:59:49.385110Z","iopub.status.idle":"2022-05-29T10:59:49.416841Z","shell.execute_reply.started":"2022-05-29T10:59:49.385081Z","shell.execute_reply":"2022-05-29T10:59:49.415840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[cat_columns].info()","metadata":{"execution":{"iopub.status.busy":"2022-05-29T10:59:49.418797Z","iopub.execute_input":"2022-05-29T10:59:49.419336Z","iopub.status.idle":"2022-05-29T10:59:49.457740Z","shell.execute_reply.started":"2022-05-29T10:59:49.419302Z","shell.execute_reply":"2022-05-29T10:59:49.456869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a = train_df['customer_ID'].unique()\nvalid = np.random.choice(a,replace=False, size=10000)\nlen(pd.unique(train_df['customer_ID'])), len(valid)\n\n#splits = (list(train_idx),list(valid_idx))","metadata":{"execution":{"iopub.status.busy":"2022-05-29T10:59:49.459258Z","iopub.execute_input":"2022-05-29T10:59:49.459819Z","iopub.status.idle":"2022-05-29T10:59:49.503846Z","shell.execute_reply.started":"2022-05-29T10:59:49.459781Z","shell.execute_reply":"2022-05-29T10:59:49.502676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_idx = train_df.index[train_df['customer_ID'].isin(valid)]\ntrain_idx = train_df.index[train_df['customer_ID'].isin(valid)==False]\n\nsplits = (list(train_idx),list(valid_idx))","metadata":{"execution":{"iopub.status.busy":"2022-05-29T10:59:49.505513Z","iopub.execute_input":"2022-05-29T10:59:49.506147Z","iopub.status.idle":"2022-05-29T10:59:49.532545Z","shell.execute_reply.started":"2022-05-29T10:59:49.506086Z","shell.execute_reply":"2022-05-29T10:59:49.531755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# DATA LOADER -- For Pytorch/fast.ai tabular learners\n\"\"\"\ndls = TabularDataLoaders.from_csv('./sample_train.csv', y_names=\"target\",\n    cat_names = cat_columns,\n    cont_names = cont_columns,\n    splits=splits,\n    procs=[Categorify, FillMissing])\n\nlen(dls.train),len(dls.valid)\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2022-05-29T10:59:49.533639Z","iopub.execute_input":"2022-05-29T10:59:49.534621Z","iopub.status.idle":"2022-05-29T10:59:49.541860Z","shell.execute_reply.started":"2022-05-29T10:59:49.534582Z","shell.execute_reply":"2022-05-29T10:59:49.540684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tabular pandas -- For fast.ai/sklearn implementation\n\nto = TabularPandas(train_df, procs=[Categorify, FillMissing],\n                   cat_names = cat_columns,\n                   cont_names = cont_columns,\n                   y_names='target',\n                   splits=splits)\n\nlen(to.train),len(to.valid)","metadata":{"execution":{"iopub.status.busy":"2022-05-29T10:59:49.543617Z","iopub.execute_input":"2022-05-29T10:59:49.545168Z","iopub.status.idle":"2022-05-29T10:59:57.598411Z","shell.execute_reply.started":"2022-05-29T10:59:49.544992Z","shell.execute_reply":"2022-05-29T10:59:57.597418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xs,y = to.train.xs,to.train.y\nvalid_xs,valid_y = to.valid.xs,to.valid.y","metadata":{"execution":{"iopub.status.busy":"2022-05-29T11:04:01.182283Z","iopub.execute_input":"2022-05-29T11:04:01.183132Z","iopub.status.idle":"2022-05-29T11:04:01.302858Z","shell.execute_reply.started":"2022-05-29T11:04:01.183073Z","shell.execute_reply":"2022-05-29T11:04:01.301698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install dtreeviz --upgrade","metadata":{"execution":{"iopub.status.busy":"2022-05-29T10:15:37.057875Z","iopub.execute_input":"2022-05-29T10:15:37.058935Z","iopub.status.idle":"2022-05-29T10:15:52.235706Z","shell.execute_reply.started":"2022-05-29T10:15:37.058880Z","shell.execute_reply":"2022-05-29T10:15:52.234274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"to.show(10)","metadata":{"execution":{"iopub.status.busy":"2022-05-29T11:04:04.793904Z","iopub.execute_input":"2022-05-29T11:04:04.794335Z","iopub.status.idle":"2022-05-29T11:04:05.138540Z","shell.execute_reply.started":"2022-05-29T11:04:04.794298Z","shell.execute_reply":"2022-05-29T11:04:05.136901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"to['target'].isna().sum()\ntrain_df['target'].isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-05-29T11:05:54.984023Z","iopub.execute_input":"2022-05-29T11:05:54.984539Z","iopub.status.idle":"2022-05-29T11:05:54.994400Z","shell.execute_reply.started":"2022-05-29T11:05:54.984501Z","shell.execute_reply":"2022-05-29T11:05:54.993235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nfrom sklearn.tree import DecisionTreeClassifier\nfrom dtreeviz.trees import *\n#from IPython.display import Image, display_svg, SVG\n\npd.options.display.max_rows = 150\npd.options.display.max_columns = 150\n\nclassifier = DecisionTreeClassifier(max_leaf_nodes=4)\nclassifier.fit(xs, y);","metadata":{"execution":{"iopub.status.busy":"2022-05-29T11:04:13.694126Z","iopub.execute_input":"2022-05-29T11:04:13.695093Z","iopub.status.idle":"2022-05-29T11:04:13.771379Z","shell.execute_reply.started":"2022-05-29T11:04:13.695052Z","shell.execute_reply":"2022-05-29T11:04:13.770239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xs.columns","metadata":{"execution":{"iopub.status.busy":"2022-05-29T10:37:31.589886Z","iopub.execute_input":"2022-05-29T10:37:31.590335Z","iopub.status.idle":"2022-05-29T10:37:31.618166Z","shell.execute_reply.started":"2022-05-29T10:37:31.590299Z","shell.execute_reply":"2022-05-29T10:37:31.616954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"viz = dtreeviz(classifier, \n               xs, \n               y,\n               target_name='default',\n               feature_names=['No','Yes'], \n               class_names=xs.columns.tolist()  # need class_names for classifier\n              )  \n              \nviz.view() ","metadata":{"execution":{"iopub.status.busy":"2022-05-29T10:38:00.477259Z","iopub.execute_input":"2022-05-29T10:38:00.478140Z","iopub.status.idle":"2022-05-29T10:38:04.956104Z","shell.execute_reply.started":"2022-05-29T10:38:00.478094Z","shell.execute_reply":"2022-05-29T10:38:04.954682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}