{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-10T18:03:33.456499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_feather('../input/parquet-files-amexdefault-prediction/train_data.ftr')\ntrain_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_feather('../input/parquet-files-amexdefault-prediction/test_data.ftr')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.drop(train_df.columns[(train_df.isnull().sum()/train_df.shape[0])*100>40],axis=1,inplace=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.drop(train_df.columns[(train_df.isnull().sum()/train_df.shape[0])*100>40],axis=1, inplace=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"[(train_df[i].isnull().sum()/train_df.shape[0])*100 for i in train_df.columns if \n ((train_df[i].isnull().sum()/train_df.shape[0])*100)>40]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.dtypes.value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.columns[train_df.dtypes=='object']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['S_2'] = pd.to_datetime(train_df['S_2'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.columns[:10]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.columns[train_df.dtypes=='float16']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfor i in train_df.columns:\n    if train_df[i].dtypes == 'float16':\n        train_df[i] = train_df[i].astype(np.float32)\n        ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.dtypes.value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['P_2'].describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"[print(i,(train_df[i].isnull().sum()/train_df.shape[0])*100) for i in train_df.columns if \n ((train_df[i].isnull().sum()/train_df.shape[0])*100)<1]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['P_2'].dropna().reset_index(drop=True).mean()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['P_2'] =train_df['P_2'].fillna(train_df['P_2'].dropna().reset_index(drop=True).mean())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col_na_1pct = [i for i in train_df.columns if \n               (i not in (['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126',\n                          'D_63', 'D_64', 'D_66', 'D_68'])) and\n ((((train_df[i].isnull().sum()/train_df.shape[0])*100)<1) and \n (((train_df[i].isnull().sum()/train_df.shape[0])*100) >0))]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(col_na_1pct)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in col_na_1pct:\n    train_df[i] =train_df[i].fillna(train_df[i].dropna().reset_index(drop=True).mean())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"[print(i,(train_df[i].isnull().sum()/train_df.shape[0])*100) for i in train_df.columns if \n ((train_df[i].isnull().sum()/train_df.shape[0])*100)<1]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['B_30'].dropna().reset_index(drop=True).mode()[0]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['B_30'] =train_df['B_30'].fillna(train_df['B_30'].dropna().reset_index(drop=True).mode()[0])\ntrain_df['B_38'] =train_df['B_38'].fillna(train_df['B_38'].dropna().reset_index(drop=True).mode()[0])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['B_30'].isnull().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"[print(i,(train_df[i].isnull().sum()/train_df.shape[0])*100) for i in train_df.columns if \n ((train_df[i].isnull().sum()/train_df.shape[0])*100)<1]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"[print(i,(train_df[i].isnull().sum()/train_df.shape[0])*100) for i in train_df.columns]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nstart = time.time()\ntrain_df.to_parquet('train_data1.parquet')\nend=time.time()\nprint(end-start)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_parquet('./train_data1.parquet')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr = train_df.corr().abs()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s= corr.unstack()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s.sort_values(ascending=False).drop_duplicates()[:40]\n#D_103, D_139, D_143,  D_119, D_75, D_74, D_79, D_48, D_107, D_48, D_115, \n# S_22, S_3,  S_15, \n#B_11, B_1, B_23,  B_2, B_14, B_16, B_33, B_12, B_3, B_30\n# R_5, R_2, R_21","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s.sort_values(ascending=False)[155:215]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.drop(['D_103','D_139','D_143','D_119', 'D_75', 'D_74', 'D_79', 'D_48' ,'D_107', 'D_115',\n              'S_22', 'S_3', 'S_15',\n              'B_11', 'B_1', 'B_23', 'B_2', 'B_14', 'B_16', 'B_33', 'B_12', 'B_3', 'B_30',\n              'R_5', 'R_2', 'R_21'], axis=1)\n#D_103, D_139, D_143,  D_119, D_75, D_74, D_79, D_48, D_107, D_115, \n# S_22, S_3,  S_15, \n#B_11, B_1, B_23,  B_2, B_14, B_16, B_33, B_12, B_3, B_30\n# R_5, R_2, R_21","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nstart = time.time()\ncorr = train_df.corr().abs()\ns= corr.unstack()\n\nend=time.time()\nprint(end-start)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s.sort_values(ascending=False).drop_duplicates()[:40]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nstart = time.time()\ntrain_df.to_parquet('train_data1.parquet')\nend=time.time()\nprint(end-start)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\ntrain_df = pd.read_parquet('../input/amex-train-data/train_data1.parquet')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.dtypes.value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['B_38'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['D_114'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['D_116'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['D_117'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['D_120'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['D_126'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['D_63'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['D_64'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['D_68'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels= pd.read_csv('../input/amex-default-prediction/train_labels.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info(max_cols=200, show_counts=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels.target.value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels.customer_ID.nunique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.customer_ID.nunique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.join(labels.set_index('customer_ID'), on='customer_ID' )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.target.value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingClassifier","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X,y = train_df.drop('target',axis=1),train_df.target","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test = train_df.drop('target',axis=1)[:4000000], train_df.drop('target',axis=1)[4000000:]\ny_train, y_test = train_df.target[:4000000], train_df.target[4000000:]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf = GradientBoostingClassifier(n_estimators=10, learning_rate=1.0,\n    max_depth=1, random_state=0).fit(X_train, y_train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf.score(X_test, y_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check the Hardware\n\n!nvidia-smi","metadata":{"execution":{"iopub.status.busy":"2022-07-10T18:48:17.186864Z","iopub.execute_input":"2022-07-10T18:48:17.187269Z","iopub.status.idle":"2022-07-10T18:48:17.913444Z","shell.execute_reply.started":"2022-07-10T18:48:17.187188Z","shell.execute_reply":"2022-07-10T18:48:17.912454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!nvcc --version","metadata":{"execution":{"iopub.status.busy":"2022-07-10T18:48:17.916100Z","iopub.execute_input":"2022-07-10T18:48:17.916807Z","iopub.status.idle":"2022-07-10T18:48:18.590419Z","shell.execute_reply.started":"2022-07-10T18:48:17.916764Z","shell.execute_reply":"2022-07-10T18:48:18.589366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport cudf\nimport cupy \nimport cuml\nimport pandas as pd\nimport matplotlib.pyplot as plt, gc, os","metadata":{"execution":{"iopub.status.busy":"2022-07-10T18:48:18.592263Z","iopub.execute_input":"2022-07-10T18:48:18.592661Z","iopub.status.idle":"2022-07-10T18:48:23.648598Z","shell.execute_reply.started":"2022-07-10T18:48:18.592596Z","shell.execute_reply":"2022-07-10T18:48:23.647788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def check_memory_usage(df):\n    start_mem = df.memory_usage().sum() / 1024 ** 2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n\nFILE_PATH_PARQUETS_train= '../input/amex-train-data/'\nFILE_PATH_PARQUETS = '../input/amex-data-integer-dtypes-parquet-format/'\nFILE_PATH_CSV = '../input/amex-default-prediction/'\n\ntrain_features = cudf.read_parquet(FILE_PATH_PARQUETS_train + 'train_data1.parquet')\ntest = cudf.read_parquet(FILE_PATH_PARQUETS + 'test.parquet')\ntrain_labels = cudf.read_csv(FILE_PATH_CSV + 'train_labels.csv')\n\ncheck_memory_usage(train_features)\ncheck_memory_usage(test)\ncheck_memory_usage(train_labels)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-10T18:48:23.650221Z","iopub.execute_input":"2022-07-10T18:48:23.650580Z","iopub.status.idle":"2022-07-10T18:49:22.388168Z","shell.execute_reply.started":"2022-07-10T18:48:23.650543Z","shell.execute_reply":"2022-07-10T18:49:22.387286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_features = ([af for af in train_features.columns])[:-1]\ncategorical_features = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\nnumerical_features = [nf for nf in all_features if nf not in categorical_features]\n\nprint(train_features.isna().sum())\nprint('================================')\nprint('Total Missing Values = {}'.format(train_features.isna().sum().sum()))\nprint('================================')","metadata":{"execution":{"iopub.status.busy":"2022-07-10T18:49:31.094378Z","iopub.execute_input":"2022-07-10T18:49:31.095203Z","iopub.status.idle":"2022-07-10T18:49:31.611510Z","shell.execute_reply.started":"2022-07-10T18:49:31.095162Z","shell.execute_reply":"2022-07-10T18:49:31.610308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_features.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T18:49:34.825408Z","iopub.execute_input":"2022-07-10T18:49:34.826354Z","iopub.status.idle":"2022-07-10T18:49:35.062219Z","shell.execute_reply.started":"2022-07-10T18:49:34.826301Z","shell.execute_reply":"2022-07-10T18:49:35.061476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_features.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-10T18:49:37.961633Z","iopub.execute_input":"2022-07-10T18:49:37.962283Z","iopub.status.idle":"2022-07-10T18:49:37.970604Z","shell.execute_reply.started":"2022-07-10T18:49:37.962245Z","shell.execute_reply":"2022-07-10T18:49:37.969502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_and_feature_engineer(df):\n    # FEATURE ENGINEERING FROM \n    # https://www.kaggle.com/code/huseyincot/amex-agg-data-how-it-created\n    all_cols = [c for c in list(df.columns) if c not in ['customer_ID','S_2']]\n    cat_features = [\"D_63\",\"D_64\"]\n    num_features = [col for col in all_cols if col not in cat_features]\n\n    test_num_agg = df.groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\n    test_num_agg.columns = ['_'.join(x) for x in test_num_agg.columns]\n\n    test_cat_agg = df.groupby(\"customer_ID\")[cat_features].agg(['count', 'last', 'nunique'])\n    test_cat_agg.columns = ['_'.join(x) for x in test_cat_agg.columns]\n\n    df = cudf.concat([test_num_agg, test_cat_agg], axis=1)\n    del test_num_agg, test_cat_agg\n    print('shape after engineering', df.shape )\n    \n    return df\n\ntrain_features = process_and_feature_engineer(train_features)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T18:49:40.977175Z","iopub.execute_input":"2022-07-10T18:49:40.977545Z","iopub.status.idle":"2022-07-10T18:49:44.968163Z","shell.execute_reply.started":"2022-07-10T18:49:40.977515Z","shell.execute_reply":"2022-07-10T18:49:44.967302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = cudf.merge(train_features, train_labels, on = 'customer_ID')\ncheck_memory_usage(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T18:49:54.069480Z","iopub.execute_input":"2022-07-10T18:49:54.070067Z","iopub.status.idle":"2022-07-10T18:49:54.285271Z","shell.execute_reply.started":"2022-07-10T18:49:54.070027Z","shell.execute_reply":"2022-07-10T18:49:54.284438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-10T18:49:57.537878Z","iopub.execute_input":"2022-07-10T18:49:57.538498Z","iopub.status.idle":"2022-07-10T18:49:57.544091Z","shell.execute_reply.started":"2022-07-10T18:49:57.538459Z","shell.execute_reply":"2022-07-10T18:49:57.543167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.to_pandas()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T18:50:09.382406Z","iopub.execute_input":"2022-07-10T18:50:09.382783Z","iopub.status.idle":"2022-07-10T18:50:15.082665Z","shell.execute_reply.started":"2022-07-10T18:50:09.382752Z","shell.execute_reply":"2022-07-10T18:50:15.081784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.concat([train,pd.get_dummies(train[['D_63_last','D_64_last']],drop_first=True)],axis=1)\ntrain.drop(['D_63_last','D_64_last'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T18:50:19.129524Z","iopub.execute_input":"2022-07-10T18:50:19.130359Z","iopub.status.idle":"2022-07-10T18:50:21.914391Z","shell.execute_reply.started":"2022-07-10T18:50:19.130320Z","shell.execute_reply":"2022-07-10T18:50:21.913419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nimport xgboost as xgb\nprint('XGB Version',xgb.__version__)\n\n# XGB MODEL PARAMETERS\nxgb_parms = { \n    'max_depth':4, \n    'learning_rate':0.05, \n    'subsample':0.8,\n    'colsample_bytree':0.6, \n    'eval_metric':'logloss',\n    'objective':'binary:logistic',\n    'tree_method':'gpu_hist',\n    'predictor':'gpu_predictor',\n    'random_state':1234\n}","metadata":{"execution":{"iopub.status.busy":"2022-07-10T18:50:24.725981Z","iopub.execute_input":"2022-07-10T18:50:24.726348Z","iopub.status.idle":"2022-07-10T18:50:24.821713Z","shell.execute_reply.started":"2022-07-10T18:50:24.726317Z","shell.execute_reply":"2022-07-10T18:50:24.820785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NEEDED WITH DeviceQuantileDMatrix BELOW\nclass IterLoadForDMatrix(xgb.core.DataIter):\n    def __init__(self, df=None, features=None, target=None, batch_size=256*1024):\n        self.features = features\n        self.target = target\n        self.df = df\n        self.it = 0 # set iterator to 0\n        self.batch_size = batch_size\n        self.batches = int( np.ceil( len(df) / self.batch_size ) )\n        super().__init__()\n\n    def reset(self):\n        '''Reset the iterator'''\n        self.it = 0\n\n    def next(self, input_data):\n        '''Yield next batch of data.'''\n        if self.it == self.batches:\n            return 0 # Return 0 when there's no more batch.\n        \n        a = self.it * self.batch_size\n        b = min( (self.it + 1) * self.batch_size, len(self.df) )\n        dt = cudf.DataFrame(self.df.iloc[a:b])\n        input_data(data=dt[self.features], label=dt[self.target]) #, weight=dt['weight'])\n        self.it += 1\n        return 1","metadata":{"execution":{"iopub.status.busy":"2022-07-10T18:50:27.594188Z","iopub.execute_input":"2022-07-10T18:50:27.594764Z","iopub.status.idle":"2022-07-10T18:50:27.603787Z","shell.execute_reply.started":"2022-07-10T18:50:27.594729Z","shell.execute_reply":"2022-07-10T18:50:27.602903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/kyakovlev\n# https://www.kaggle.com/competitions/amex-default-prediction/discussion/327534\ndef amex_metric_mod(y_true, y_pred):\n\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    return 0.5 * (gini[1]/gini[0] + top_four)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T18:50:29.015276Z","iopub.execute_input":"2022-07-10T18:50:29.016023Z","iopub.status.idle":"2022-07-10T18:50:29.026180Z","shell.execute_reply.started":"2022-07-10T18:50:29.015985Z","shell.execute_reply":"2022-07-10T18:50:29.025054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"importances = []\noof = []\n# train = train.to_pandas() # free GPU memory\nTRAIN_SUBSAMPLE = 0.8\nimport gc\ngc.collect()\nSEED=1234\nfolds=3\nFEATURES = train.drop(['customer_ID','target'],axis=1).columns\nVER=1\n\n\nskf = KFold(n_splits=folds, shuffle=True, random_state=1234)\nfor fold,(train_idx, valid_idx) in enumerate(skf.split(\n            train, train.target )):\n    \n    # TRAIN WITH SUBSAMPLE OF TRAIN FOLD DATA\n    if TRAIN_SUBSAMPLE<1.0:\n        np.random.seed(SEED)\n        train_idx = np.random.choice(train_idx, \n                       int(len(train_idx)*TRAIN_SUBSAMPLE), replace=False)\n        np.random.seed(None)\n    \n    print('#'*25)\n    print('### Fold',fold+1)\n    print('### Train size',len(train_idx),'Valid size',len(valid_idx))\n    print(f'### Training with {int(TRAIN_SUBSAMPLE*100)}% fold data...')\n    print('#'*25)\n    \n    # TRAIN, VALID, TEST FOR FOLD K\n    Xy_train = IterLoadForDMatrix(train.loc[train_idx], FEATURES, 'target')\n    X_valid = train.loc[valid_idx, FEATURES]\n    y_valid = train.loc[valid_idx, 'target']\n    \n    dtrain = xgb.DeviceQuantileDMatrix(Xy_train, max_bin=256)\n    dvalid = xgb.DMatrix(data=X_valid, label=y_valid)\n    \n    # TRAIN MODEL FOLD K\n    model = xgb.train(xgb_parms, \n                dtrain=dtrain,\n                evals=[(dtrain,'train'),(dvalid,'valid')],\n                num_boost_round=9999,\n                early_stopping_rounds=50,\n                verbose_eval=100) \n    model.save_model(f'XGB_v{VER}_fold{fold}.xgb')\n    \n    # GET FEATURE IMPORTANCE FOR FOLD K\n    dd = model.get_score(importance_type='weight')\n    df = pd.DataFrame({'feature':dd.keys(),f'importance_{fold}':dd.values()})\n    importances.append(df)\n            \n    # INFER OOF FOLD K\n    oof_preds = model.predict(dvalid)\n    acc = amex_metric_mod(y_valid.values, oof_preds)\n    print('Kaggle Metric =',acc,'\\n')\n    \n    # SAVE OOF\n    df = train.loc[valid_idx, ['customer_ID','target'] ].copy()\n    df['oof_pred'] = oof_preds\n    oof.append( df )\n    \n    del dtrain, Xy_train, dd, df\n    del X_valid, y_valid, dvalid, model\n    _ = gc.collect()\n    \nprint('#'*25)\noof = pd.concat(oof,axis=0,ignore_index=True).set_index('customer_ID')\nacc = amex_metric_mod(oof.target.values, oof.oof_pred.values)\nprint('OVERALL CV Kaggle Metric =',acc)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T18:50:31.395781Z","iopub.execute_input":"2022-07-10T18:50:31.396410Z","iopub.status.idle":"2022-07-10T18:52:42.431249Z","shell.execute_reply.started":"2022-07-10T18:50:31.396374Z","shell.execute_reply":"2022-07-10T18:52:42.430321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2022-07-10T18:52:47.790277Z","iopub.execute_input":"2022-07-10T18:52:47.790658Z","iopub.status.idle":"2022-07-10T18:52:47.910896Z","shell.execute_reply.started":"2022-07-10T18:52:47.790599Z","shell.execute_reply":"2022-07-10T18:52:47.909929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2022-07-10T18:52:52.583050Z","iopub.execute_input":"2022-07-10T18:52:52.583406Z","iopub.status.idle":"2022-07-10T18:52:52.868361Z","shell.execute_reply.started":"2022-07-10T18:52:52.583375Z","shell.execute_reply":"2022-07-10T18:52:52.867537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_xgb = pd.read_parquet(FILE_PATH_PARQUETS_train, columns=['customer_ID']).drop_duplicates()\noof_xgb['customer_ID_hash'] = oof_xgb['customer_ID'].apply(lambda x: int(x[-16:],16) ).astype('int64')\noof_xgb = oof_xgb.set_index('customer_ID_hash')\noof_xgb = oof_xgb.merge(oof, left_index=True, right_index=True)\noof_xgb = oof_xgb.sort_index().reset_index(drop=True)\noof_xgb.to_csv(f'oof_xgb_v{VER}.csv',index=False)\noof_xgb.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T18:45:04.609865Z","iopub.execute_input":"2022-07-10T18:45:04.610488Z","iopub.status.idle":"2022-07-10T18:45:07.453732Z","shell.execute_reply.started":"2022-07-10T18:45:04.610445Z","shell.execute_reply":"2022-07-10T18:45:07.452771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = importances[0].copy()\nfor k in range(1,folds): df = df.merge(importances[k], on='feature', how='left')\ndf['importance'] = df.iloc[:,1:].mean(axis=1)\ndf = df.sort_values('importance',ascending=False)\nNUM_FEATURES = 20\nplt.figure(figsize=(10,5*NUM_FEATURES//10))\nplt.barh(np.arange(NUM_FEATURES,0,-1), df.importance.values[:NUM_FEATURES])\nplt.yticks(np.arange(NUM_FEATURES,0,-1), df.feature.values[:NUM_FEATURES])\nplt.title(f'XGB Feature Importance - Top {NUM_FEATURES}')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T18:45:07.455097Z","iopub.execute_input":"2022-07-10T18:45:07.455551Z","iopub.status.idle":"2022-07-10T18:45:07.772254Z","shell.execute_reply.started":"2022-07-10T18:45:07.455512Z","shell.execute_reply":"2022-07-10T18:45:07.771371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CALCULATE SIZE OF EACH SEPARATE TEST PART\ndef get_rows(customers, test, NUM_PARTS = 4, verbose = ''):\n    chunk = len(customers)//NUM_PARTS\n    if verbose != '':\n        print(f'We will process {verbose} data as {NUM_PARTS} separate parts.')\n        print(f'There will be {chunk} customers in each part (except the last part).')\n        print('Below are number of rows in each part:')\n    rows = []\n\n    for k in range(NUM_PARTS):\n        if k==NUM_PARTS-1: cc = customers[k*chunk:]\n        else: cc = customers[k*chunk:(k+1)*chunk]\n        s = test.loc[test.customer_ID.isin(cc)].shape[0]\n        rows.append(s)\n    if verbose != '': print( rows )\n    return rows,chunk\n\n# COMPUTE SIZE OF 4 PARTS FOR TEST DATA\nNUM_PARTS = 4\nFILE_PATH_PARQUETS = '../input/amex-data-integer-dtypes-parquet-format/'\n\nprint(f'Reading test data...')\ntest = cudf.read_parquet(FILE_PATH_PARQUETS + 'test.parquet')\n# customers = test[['customer_ID']].drop_duplicates().sort_index().values.flatten()\n# rows,num_cust = get_rows(customers, test[['customer_ID']], NUM_PARTS = NUM_PARTS, verbose = 'test')","metadata":{"execution":{"iopub.status.busy":"2022-07-10T18:47:01.281538Z","iopub.execute_input":"2022-07-10T18:47:01.282531Z","iopub.status.idle":"2022-07-10T18:47:03.261142Z","shell.execute_reply.started":"2022-07-10T18:47:01.282488Z","shell.execute_reply":"2022-07-10T18:47:03.259769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom catboost import CatBoostClassifier\nfrom sklearn.model_selection import StratifiedKFold","metadata":{"execution":{"iopub.status.busy":"2022-07-19T19:15:29.505779Z","iopub.execute_input":"2022-07-19T19:15:29.506223Z","iopub.status.idle":"2022-07-19T19:15:30.774380Z","shell.execute_reply.started":"2022-07-19T19:15:29.506135Z","shell.execute_reply":"2022-07-19T19:15:30.773478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_pickle(\"../input/amex-agg-data-pickle/train_agg.pkl\", compression=\"gzip\")\ntest = pd.read_pickle(\"../input/amex-agg-data-pickle/test_agg.pkl\", compression=\"gzip\")","metadata":{"execution":{"iopub.status.busy":"2022-07-19T19:15:30.778449Z","iopub.execute_input":"2022-07-19T19:15:30.778804Z","iopub.status.idle":"2022-07-19T19:16:28.851089Z","shell.execute_reply.started":"2022-07-19T19:15:30.778771Z","shell.execute_reply":"2022-07-19T19:16:28.850092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T19:16:28.852319Z","iopub.execute_input":"2022-07-19T19:16:28.852727Z","iopub.status.idle":"2022-07-19T19:16:29.101450Z","shell.execute_reply.started":"2022-07-19T19:16:28.852669Z","shell.execute_reply":"2022-07-19T19:16:29.100612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x == 0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n\n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x == 0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T19:16:29.103410Z","iopub.execute_input":"2022-07-19T19:16:29.103974Z","iopub.status.idle":"2022-07-19T19:16:29.118452Z","shell.execute_reply.started":"2022-07-19T19:16:29.103930Z","shell.execute_reply":"2022-07-19T19:16:29.117630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T19:16:29.119788Z","iopub.execute_input":"2022-07-19T19:16:29.120556Z","iopub.status.idle":"2022-07-19T19:16:30.823323Z","shell.execute_reply.started":"2022-07-19T19:16:29.120516Z","shell.execute_reply":"2022-07-19T19:16:30.822265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['B_38_last'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T19:20:32.250625Z","iopub.execute_input":"2022-07-18T19:20:32.252092Z","iopub.status.idle":"2022-07-18T19:20:32.270391Z","shell.execute_reply.started":"2022-07-18T19:20:32.252050Z","shell.execute_reply":"2022-07-18T19:20:32.269651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['D_63_last'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T19:20:32.271671Z","iopub.execute_input":"2022-07-18T19:20:32.272111Z","iopub.status.idle":"2022-07-18T19:20:32.283146Z","shell.execute_reply.started":"2022-07-18T19:20:32.272067Z","shell.execute_reply":"2022-07-18T19:20:32.282278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.dtypes.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T19:20:32.284327Z","iopub.execute_input":"2022-07-18T19:20:32.284713Z","iopub.status.idle":"2022-07-18T19:20:32.295356Z","shell.execute_reply.started":"2022-07-18T19:20:32.284677Z","shell.execute_reply":"2022-07-18T19:20:32.294484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.columns[train.dtypes=='category'].values.tolist()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T19:20:32.296646Z","iopub.execute_input":"2022-07-18T19:20:32.297110Z","iopub.status.idle":"2022-07-18T19:20:32.311187Z","shell.execute_reply.started":"2022-07-18T19:20:32.297075Z","shell.execute_reply":"2022-07-18T19:20:32.310275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = test.columns.to_list()\ncat_features = [\n    \"B_30\",\n    \"B_38\",\n    \"D_114\",\n    \"D_116\",\n    \"D_117\",\n    \"D_120\",\n    \"D_126\",\n    \"D_66\",\n    \"D_68\"\n]\ncat_features = [f\"{cf}_last\" for cf in cat_features]","metadata":{"execution":{"iopub.status.busy":"2022-07-18T19:20:32.314309Z","iopub.execute_input":"2022-07-18T19:20:32.314708Z","iopub.status.idle":"2022-07-18T19:20:32.321279Z","shell.execute_reply.started":"2022-07-18T19:20:32.314684Z","shell.execute_reply":"2022-07-18T19:20:32.320499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.concat([train,pd.get_dummies(train[['D_63_last', 'D_64_last']])],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T19:20:32.323391Z","iopub.execute_input":"2022-07-18T19:20:32.323677Z","iopub.status.idle":"2022-07-18T19:20:38.953904Z","shell.execute_reply.started":"2022-07-18T19:20:32.323651Z","shell.execute_reply":"2022-07-18T19:20:38.953065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop(['D_63_last', 'D_64_last'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T19:20:38.955177Z","iopub.execute_input":"2022-07-18T19:20:38.955695Z","iopub.status.idle":"2022-07-18T19:20:40.718222Z","shell.execute_reply.started":"2022-07-18T19:20:38.955657Z","shell.execute_reply":"2022-07-18T19:20:40.717399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = train.drop('target',axis=1)\ny_train = train['target']","metadata":{"execution":{"iopub.status.busy":"2022-07-18T19:20:40.719529Z","iopub.execute_input":"2022-07-18T19:20:40.720046Z","iopub.status.idle":"2022-07-18T19:20:42.001840Z","shell.execute_reply.started":"2022-07-18T19:20:40.720008Z","shell.execute_reply":"2022-07-18T19:20:42.001006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-18T19:20:42.003130Z","iopub.execute_input":"2022-07-18T19:20:42.003514Z","iopub.status.idle":"2022-07-18T19:20:42.009308Z","shell.execute_reply.started":"2022-07-18T19:20:42.003476Z","shell.execute_reply":"2022-07-18T19:20:42.008266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.shape[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-18T19:20:42.010900Z","iopub.execute_input":"2022-07-18T19:20:42.011355Z","iopub.status.idle":"2022-07-18T19:20:42.022483Z","shell.execute_reply.started":"2022-07-18T19:20:42.011310Z","shell.execute_reply":"2022-07-18T19:20:42.021467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-07-18T19:20:42.024061Z","iopub.execute_input":"2022-07-18T19:20:42.024442Z","iopub.status.idle":"2022-07-18T19:20:42.033245Z","shell.execute_reply.started":"2022-07-18T19:20:42.024404Z","shell.execute_reply":"2022-07-18T19:20:42.032484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train1, X_test, y_train1, y_test = train_test_split(X_train, y_train, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T19:20:42.034909Z","iopub.execute_input":"2022-07-18T19:20:42.035521Z","iopub.status.idle":"2022-07-18T19:20:47.081523Z","shell.execute_reply.started":"2022-07-18T19:20:42.035485Z","shell.execute_reply":"2022-07-18T19:20:47.080306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf = CatBoostClassifier(iterations=50, random_state=22)\nclf.fit(X_train1, y_train1, eval_set=[(X_test, y_test)],  verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T19:25:06.125513Z","iopub.execute_input":"2022-07-18T19:25:06.125900Z","iopub.status.idle":"2022-07-18T19:28:33.030044Z","shell.execute_reply.started":"2022-07-18T19:25:06.125867Z","shell.execute_reply":"2022-07-18T19:28:33.029250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf.predict_proba(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T19:30:40.261604Z","iopub.execute_input":"2022-07-18T19:30:40.262433Z","iopub.status.idle":"2022-07-18T19:31:05.318536Z","shell.execute_reply.started":"2022-07-18T19:30:40.262396Z","shell.execute_reply":"2022-07-18T19:31:05.317694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_oof = np.zeros(X_train.shape[0])\ny_oof[100]","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:29:13.730469Z","iopub.execute_input":"2022-07-11T20:29:13.730836Z","iopub.status.idle":"2022-07-11T20:29:13.738690Z","shell.execute_reply.started":"2022-07-11T20:29:13.730805Z","shell.execute_reply":"2022-07-11T20:29:13.737617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N_FOLDS = 5\nskf = StratifiedKFold(n_splits=N_FOLDS, shuffle=True, random_state=22)\nfor train_ind, val_ind in skf.split(X_train, y_train):\n    print(train_ind, val_ind)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:30:13.053821Z","iopub.execute_input":"2022-07-11T20:30:13.054549Z","iopub.status.idle":"2022-07-11T20:30:13.127645Z","shell.execute_reply.started":"2022-07-11T20:30:13.054507Z","shell.execute_reply":"2022-07-11T20:30:13.126720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N_FOLDS = 5\nskf = StratifiedKFold(n_splits=N_FOLDS, shuffle=True, random_state=22)\ny_oof = np.zeros(X_train.shape[0])\ny_test = np.zeros(test.shape[0])\nix = 0\nfor train_ind, val_ind in skf.split(X_train, y_train):\n    print(f\"******* Fold {ix} ******* \")\n    tr_x, val_x = (\n        X_train.iloc[train_ind].reset_index(drop=True),\n        X_train.iloc[val_ind].reset_index(drop=True),\n    )\n    tr_y, val_y = (\n        y_train.iloc[train_ind].reset_index(drop=True),\n        y_train.iloc[val_ind].reset_index(drop=True),\n    )\n\n    clf = CatBoostClassifier(iterations=5000, random_state=22)\n    clf.fit(tr_x, tr_y, eval_set=[(val_x, val_y)], cat_features=cat_features,  verbose=100)\n    preds = clf.predict_proba(val_x)[:, 1]\n    y_oof[val_ind] = y_oof[val_ind] + preds\n\n    preds_test = clf.predict_proba(test)[:, 1]\n    y_test = y_test + preds_test / N_FOLDS\n    ix = ix + 1\ny_pred = train_y.copy(deep=True)\ny_pred = y_pred.rename(columns={\"target\": \"prediction\"})\ny_pred[\"prediction\"] = y_oof\nval_score = amex_metric(train_y, y_pred)\nprint(f\"Amex metric: {val_score}\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LOAD LIBRARIES\nimport pandas as pd, numpy as np # CPU libraries\nimport cupy, cudf # GPU libraries\nimport matplotlib.pyplot as plt, gc, os\n\nprint('RAPIDS version',cudf.__version__)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:00:25.582911Z","iopub.execute_input":"2022-07-28T18:00:25.583359Z","iopub.status.idle":"2022-07-28T18:00:30.405421Z","shell.execute_reply.started":"2022-07-28T18:00:25.583276Z","shell.execute_reply":"2022-07-28T18:00:30.404529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# VERSION NAME FOR SAVED MODEL FILES\nVER = 1\n\n# TRAIN RANDOM SEED\nSEED = 42\n\n# FILL NAN VALUE\nNAN_VALUE = -127 # will fit in int8\n\n# FOLDS PER MODEL\nFOLDS = 5","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:00:30.407425Z","iopub.execute_input":"2022-07-28T18:00:30.408153Z","iopub.status.idle":"2022-07-28T18:00:30.414267Z","shell.execute_reply.started":"2022-07-28T18:00:30.408097Z","shell.execute_reply":"2022-07-28T18:00:30.413188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_file(path = '', usecols = None):\n    # LOAD DATAFRAME\n    if usecols is not None: df = cudf.read_parquet(path, columns=usecols)\n    else: df = cudf.read_parquet(path)\n    # REDUCE DTYPE FOR CUSTOMER AND DATE\n    df['customer_ID'] = df['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    df.S_2 = cudf.to_datetime( df.S_2 )\n    # SORT BY CUSTOMER AND DATE (so agg('last') works correctly)\n    #df = df.sort_values(['customer_ID','S_2'])\n    #df = df.reset_index(drop=True)\n    # FILL NAN\n    df = df.fillna(NAN_VALUE) \n    print('shape of data:', df.shape)\n    \n    return df\n\nprint('Reading train data...')\nTRAIN_PATH = '../input/amex-data-integer-dtypes-parquet-format/train.parquet'\ntrain = read_file(path = TRAIN_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:00:30.416494Z","iopub.execute_input":"2022-07-28T18:00:30.418287Z","iopub.status.idle":"2022-07-28T18:00:54.750615Z","shell.execute_reply.started":"2022-07-28T18:00:30.418240Z","shell.execute_reply":"2022-07-28T18:00:54.749790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_and_feature_engineer(df):\n    # FEATURE ENGINEERING FROM \n    # https://www.kaggle.com/code/huseyincot/amex-agg-data-how-it-created\n    all_cols = [c for c in list(df.columns) if c not in ['customer_ID','S_2']]\n    cat_features = [\"B_30\",\"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\n    num_features = [col for col in all_cols if col not in cat_features]\n\n    test_num_agg = df.groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\n    test_num_agg.columns = ['_'.join(x) for x in test_num_agg.columns]\n\n    test_cat_agg = df.groupby(\"customer_ID\")[cat_features].agg(['count', 'last', 'nunique'])\n    test_cat_agg.columns = ['_'.join(x) for x in test_cat_agg.columns]\n\n    df = cudf.concat([test_num_agg, test_cat_agg], axis=1)\n    del test_num_agg, test_cat_agg\n    print('shape after engineering', df.shape )\n    \n    return df\n\ntrain = process_and_feature_engineer(train)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:00:54.752351Z","iopub.execute_input":"2022-07-28T18:00:54.754038Z","iopub.status.idle":"2022-07-28T18:00:56.046517Z","shell.execute_reply.started":"2022-07-28T18:00:54.753992Z","shell.execute_reply":"2022-07-28T18:00:56.045690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ADD TARGETS\ntargets = cudf.read_csv('../input/amex-default-prediction/train_labels.csv')\ntargets['customer_ID'] = targets['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\ntargets = targets.set_index('customer_ID')\ntrain = train.merge(targets, left_index=True, right_index=True, how='left')\ntrain.target = train.target.astype('int8')\ndel targets\n\n# NEEDED TO MAKE CV DETERMINISTIC (cudf merge above randomly shuffles rows)\ntrain = train.sort_index().reset_index()\n\n# FEATURES\nFEATURES = train.columns[1:-1]\nprint(f'There are {len(FEATURES)} features!')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:01:00.630704Z","iopub.execute_input":"2022-07-28T18:01:00.631272Z","iopub.status.idle":"2022-07-28T18:01:02.583581Z","shell.execute_reply.started":"2022-07-28T18:01:00.631228Z","shell.execute_reply":"2022-07-28T18:01:02.582717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LOAD XGB LIBRARY\nfrom sklearn.model_selection import KFold\nimport xgboost as xgb\nprint('XGB Version',xgb.__version__)\n\n# XGB MODEL PARAMETERS\nxgb_parms = { \n    'max_depth':4, \n    'learning_rate':0.05, \n    'subsample':0.8,\n    'colsample_bytree':0.6, \n    'eval_metric':'logloss',\n    'objective':'binary:logistic',\n    'tree_method':'gpu_hist',\n    'predictor':'gpu_predictor',\n    'random_state':SEED\n}","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:01:19.500392Z","iopub.execute_input":"2022-07-28T18:01:19.500782Z","iopub.status.idle":"2022-07-28T18:01:19.590134Z","shell.execute_reply.started":"2022-07-28T18:01:19.500744Z","shell.execute_reply":"2022-07-28T18:01:19.589309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NEEDED WITH DeviceQuantileDMatrix BELOW\nclass IterLoadForDMatrix(xgb.core.DataIter):\n    def __init__(self, df=None, features=None, target=None, batch_size=256*1024):\n        self.features = features\n        self.target = target\n        self.df = df\n        self.it = 0 # set iterator to 0\n        self.batch_size = batch_size\n        self.batches = int( np.ceil( len(df) / self.batch_size ) )\n        super().__init__()\n\n    def reset(self):\n        '''Reset the iterator'''\n        self.it = 0\n\n    def next(self, input_data):\n        '''Yield next batch of data.'''\n        if self.it == self.batches:\n            return 0 # Return 0 when there's no more batch.\n        \n        a = self.it * self.batch_size\n        b = min( (self.it + 1) * self.batch_size, len(self.df) )\n        dt = cudf.DataFrame(self.df.iloc[a:b])\n        input_data(data=dt[self.features], label=dt[self.target]) #, weight=dt['weight'])\n        self.it += 1\n        return 1\n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:01:44.791842Z","iopub.execute_input":"2022-07-28T18:01:44.792465Z","iopub.status.idle":"2022-07-28T18:01:44.801300Z","shell.execute_reply.started":"2022-07-28T18:01:44.792428Z","shell.execute_reply":"2022-07-28T18:01:44.800528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric_mod(y_true, y_pred):\n\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    return 0.5 * (gini[1]/gini[0] + top_four)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:02:11.905277Z","iopub.execute_input":"2022-07-28T18:02:11.906247Z","iopub.status.idle":"2022-07-28T18:02:11.919746Z","shell.execute_reply.started":"2022-07-28T18:02:11.906199Z","shell.execute_reply":"2022-07-28T18:02:11.918687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"importances = []\noof = []\ntrain = train.to_pandas() # free GPU memory\nTRAIN_SUBSAMPLE = 1.0\ngc.collect()\n\nskf = KFold(n_splits=FOLDS, shuffle=True, random_state=SEED)\nfor fold,(train_idx, valid_idx) in enumerate(skf.split(\n            train, train.target )):\n    \n    # TRAIN WITH SUBSAMPLE OF TRAIN FOLD DATA\n    if TRAIN_SUBSAMPLE<1.0:\n        np.random.seed(SEED)\n        train_idx = np.random.choice(train_idx, \n                       int(len(train_idx)*TRAIN_SUBSAMPLE), replace=False)\n        np.random.seed(None)\n    \n    print('#'*25)\n    print('### Fold',fold+1)\n    print('### Train size',len(train_idx),'Valid size',len(valid_idx))\n    print(f'### Training with {int(TRAIN_SUBSAMPLE*100)}% fold data...')\n    print('#'*25)\n    \n    # TRAIN, VALID, TEST FOR FOLD K\n    Xy_train = IterLoadForDMatrix(train.loc[train_idx], FEATURES, 'target')\n    X_valid = train.loc[valid_idx, FEATURES]\n    y_valid = train.loc[valid_idx, 'target']\n    \n    dtrain = xgb.DeviceQuantileDMatrix(Xy_train, max_bin=256)\n    dvalid = xgb.DMatrix(data=X_valid, label=y_valid)\n    \n    # TRAIN MODEL FOLD K\n    model = xgb.train(xgb_parms, \n                dtrain=dtrain,\n                evals=[(dtrain,'train'),(dvalid,'valid')],\n                num_boost_round=9999,\n                early_stopping_rounds=100,\n                verbose_eval=100) \n    model.save_model(f'XGB_v{VER}_fold{fold}.xgb')\n    \n    # GET FEATURE IMPORTANCE FOR FOLD K\n    dd = model.get_score(importance_type='weight')\n    df = pd.DataFrame({'feature':dd.keys(),f'importance_{fold}':dd.values()})\n    importances.append(df)\n            \n    # INFER OOF FOLD K\n    oof_preds = model.predict(dvalid)\n    acc = amex_metric_mod(y_valid.values, oof_preds)\n    print('Kaggle Metric =',acc,'\\n')\n    \n    # SAVE OOF\n    df = train.loc[valid_idx, ['customer_ID','target'] ].copy()\n    df['oof_pred'] = oof_preds\n    oof.append( df )\n    \n    del dtrain, Xy_train, dd, df\n    del X_valid, y_valid, dvalid, model\n    _ = gc.collect()\n    \nprint('#'*25)\noof = pd.concat(oof,axis=0,ignore_index=True).set_index('customer_ID')\nacc = amex_metric_mod(oof.target.values, oof.oof_pred.values)\nprint('OVERALL CV Kaggle Metric =',acc)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:02:39.645360Z","iopub.execute_input":"2022-07-28T18:02:39.645704Z","iopub.status.idle":"2022-07-28T18:11:57.821019Z","shell.execute_reply.started":"2022-07-28T18:02:39.645673Z","shell.execute_reply":"2022-07-28T18:11:57.820063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CLEAN RAM\ndel train\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:11:57.822681Z","iopub.execute_input":"2022-07-28T18:11:57.824020Z","iopub.status.idle":"2022-07-28T18:11:57.961569Z","shell.execute_reply.started":"2022-07-28T18:11:57.823973Z","shell.execute_reply":"2022-07-28T18:11:57.960656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_xgb = pd.read_parquet(TRAIN_PATH, columns=['customer_ID']).drop_duplicates()\noof_xgb['customer_ID_hash'] = oof_xgb['customer_ID'].apply(lambda x: int(x[-16:],16) ).astype('int64')\noof_xgb = oof_xgb.set_index('customer_ID_hash')\noof_xgb = oof_xgb.merge(oof, left_index=True, right_index=True)\noof_xgb = oof_xgb.sort_index().reset_index(drop=True)\noof_xgb.to_csv(f'oof_xgb_v{VER}.csv',index=False)\noof_xgb.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:11:57.963124Z","iopub.execute_input":"2022-07-28T18:11:57.963612Z","iopub.status.idle":"2022-07-28T18:12:03.458306Z","shell.execute_reply.started":"2022-07-28T18:11:57.963568Z","shell.execute_reply":"2022-07-28T18:12:03.457554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PLOT OOF PREDICTIONS\nplt.hist(oof_xgb.oof_pred.values, bins=100)\nplt.title('OOF Predictions')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:12:03.460336Z","iopub.execute_input":"2022-07-28T18:12:03.460911Z","iopub.status.idle":"2022-07-28T18:12:03.843504Z","shell.execute_reply.started":"2022-07-28T18:12:03.460873Z","shell.execute_reply":"2022-07-28T18:12:03.842700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CLEAR VRAM, RAM FOR INFERENCE BELOW\ndel oof_xgb, oof\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:12:03.844985Z","iopub.execute_input":"2022-07-28T18:12:03.845569Z","iopub.status.idle":"2022-07-28T18:12:04.054248Z","shell.execute_reply.started":"2022-07-28T18:12:03.845528Z","shell.execute_reply":"2022-07-28T18:12:04.052939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\ndf = importances[0].copy()\nfor k in range(1,FOLDS): df = df.merge(importances[k], on='feature', how='left')\ndf['importance'] = df.iloc[:,1:].mean(axis=1)\ndf = df.sort_values('importance',ascending=False)\ndf.to_csv(f'xgb_feature_importance_v{VER}.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:12:04.060249Z","iopub.execute_input":"2022-07-28T18:12:04.060788Z","iopub.status.idle":"2022-07-28T18:12:04.103563Z","shell.execute_reply.started":"2022-07-28T18:12:04.060747Z","shell.execute_reply":"2022-07-28T18:12:04.102780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUM_FEATURES = 20\nplt.figure(figsize=(10,5*NUM_FEATURES//10))\nplt.barh(np.arange(NUM_FEATURES,0,-1), df.importance.values[:NUM_FEATURES])\nplt.yticks(np.arange(NUM_FEATURES,0,-1), df.feature.values[:NUM_FEATURES])\nplt.title(f'XGB Feature Importance - Top {NUM_FEATURES}')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:12:04.108163Z","iopub.execute_input":"2022-07-28T18:12:04.110464Z","iopub.status.idle":"2022-07-28T18:12:04.471417Z","shell.execute_reply.started":"2022-07-28T18:12:04.110425Z","shell.execute_reply":"2022-07-28T18:12:04.470680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CALCULATE SIZE OF EACH SEPARATE TEST PART\ndef get_rows(customers, test, NUM_PARTS = 4, verbose = ''):\n    chunk = len(customers)//NUM_PARTS\n    if verbose != '':\n        print(f'We will process {verbose} data as {NUM_PARTS} separate parts.')\n        print(f'There will be {chunk} customers in each part (except the last part).')\n        print('Below are number of rows in each part:')\n    rows = []\n\n    for k in range(NUM_PARTS):\n        if k==NUM_PARTS-1: cc = customers[k*chunk:]\n        else: cc = customers[k*chunk:(k+1)*chunk]\n        s = test.loc[test.customer_ID.isin(cc)].shape[0]\n        rows.append(s)\n    if verbose != '': print( rows )\n    return rows,chunk\n\n# COMPUTE SIZE OF 4 PARTS FOR TEST DATA\nNUM_PARTS = 4\nTEST_PATH = '../input/amex-data-integer-dtypes-parquet-format/test.parquet'\n\nprint(f'Reading test data...')\ntest = read_file(path = TEST_PATH, usecols = ['customer_ID','S_2'])\ncustomers = test[['customer_ID']].drop_duplicates().sort_index().values.flatten()\nrows,num_cust = get_rows(customers, test[['customer_ID']], NUM_PARTS = NUM_PARTS, verbose = 'test')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:12:04.472956Z","iopub.execute_input":"2022-07-28T18:12:04.473538Z","iopub.status.idle":"2022-07-28T18:12:07.369931Z","shell.execute_reply.started":"2022-07-28T18:12:04.473500Z","shell.execute_reply":"2022-07-28T18:12:07.369122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# INFER TEST DATA IN PARTS\nskip_rows = 0\nskip_cust = 0\ntest_preds = []\n\nfor k in range(NUM_PARTS):\n    \n    # READ PART OF TEST DATA\n    print(f'\\nReading test data...')\n    test = read_file(path = TEST_PATH)\n    test = test.iloc[skip_rows:skip_rows+rows[k]]\n    skip_rows += rows[k]\n    print(f'=> Test part {k+1} has shape', test.shape )\n    \n    # PROCESS AND FEATURE ENGINEER PART OF TEST DATA\n    test = process_and_feature_engineer(test)\n    if k==NUM_PARTS-1: test = test.loc[customers[skip_cust:]]\n    else: test = test.loc[customers[skip_cust:skip_cust+num_cust]]\n    skip_cust += num_cust\n    \n    # TEST DATA FOR XGB\n    X_test = test[FEATURES]\n    dtest = xgb.DMatrix(data=X_test)\n    test = test[['P_2_mean']] # reduce memory\n    del X_test\n    gc.collect()\n\n    # INFER XGB MODELS ON TEST DATA\n    model = xgb.Booster()\n    model.load_model(f'XGB_v{VER}_fold0.xgb')\n    preds = model.predict(dtest)\n    for f in range(1,FOLDS):\n        model.load_model(f'XGB_v{VER}_fold{f}.xgb')\n        preds += model.predict(dtest)\n    preds /= FOLDS\n    test_preds.append(preds)\n\n    # CLEAN MEMORY\n    del dtest, model\n    _ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:12:07.371270Z","iopub.execute_input":"2022-07-28T18:12:07.371782Z","iopub.status.idle":"2022-07-28T18:14:36.557905Z","shell.execute_reply.started":"2022-07-28T18:12:07.371743Z","shell.execute_reply":"2022-07-28T18:14:36.557043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# WRITE SUBMISSION FILE\ntest_preds = np.concatenate(test_preds)\ntest = cudf.DataFrame(index=customers,data={'prediction':test_preds})\nsub = cudf.read_csv('../input/amex-default-prediction/sample_submission.csv')[['customer_ID']]\nsub['customer_ID_hash'] = sub['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\nsub = sub.set_index('customer_ID_hash')\nsub = sub.merge(test[['prediction']], left_index=True, right_index=True, how='left')\nsub = sub.reset_index(drop=True)\n\n# DISPLAY PREDICTIONS\nsub.to_csv(f'submission_xgb_v{VER}.csv',index=False)\nprint('Submission file shape is', sub.shape )\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:14:36.560376Z","iopub.execute_input":"2022-07-28T18:14:36.560661Z","iopub.status.idle":"2022-07-28T18:14:37.673796Z","shell.execute_reply.started":"2022-07-28T18:14:36.560635Z","shell.execute_reply":"2022-07-28T18:14:37.672987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PLOT PREDICTIONS\nplt.hist(sub.to_pandas().prediction, bins=100)\nplt.title('Test Predictions')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:14:37.675261Z","iopub.execute_input":"2022-07-28T18:14:37.675921Z","iopub.status.idle":"2022-07-28T18:14:38.644330Z","shell.execute_reply.started":"2022-07-28T18:14:37.675878Z","shell.execute_reply":"2022-07-28T18:14:38.643546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}