{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Amex 💳 TabTransformer 🤖⚔️🤖\n\nAn attempt at TabTransformer with a few minor changes from Keras impln \n\n**Paper**\nhttps://arxiv.org/pdf/2012.06678.pdf\n\n**Credits**\n* Feature engg tips from [Ambrosm](https://www.kaggle.com/ambrosm) and others\n* NN pipeline tips from [Ambrosm](https://www.kaggle.com/ambrosm)  notebook\n\n\n**Note**\n- I have not shared my full feature engg here and mostly presented how Tab Transformer could be implements in the spirit of the competition .\n- Not Added Infer code as mostly was running this on Collab. But it should not be hard to add .\n- Used less features so it completes in Kaggle , Personally run with many more on Collab","metadata":{"id":"6S1tGBZ5grxs"}},{"cell_type":"markdown","source":"### Pre-requisites","metadata":{"id":"YsemIH1bhEV-"}},{"cell_type":"code","source":"%%bash \npip install colorama --quiet \npip install tensorflow_addons --quiet \npip install adabelief-tf --no-cache-dir  --quiet ","metadata":{"id":"n_LR4Mhdm3QH","jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-08-09T23:18:31.170439Z","iopub.execute_input":"2022-08-09T23:18:31.171325Z","iopub.status.idle":"2022-08-09T23:18:59.581828Z","shell.execute_reply.started":"2022-08-09T23:18:31.171217Z","shell.execute_reply":"2022-08-09T23:18:59.580420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Imports and Constants","metadata":{"id":"_RMauooTHaLw"}},{"cell_type":"code","source":"import os,sys,random\nos.environ['CUDA_LAUNCH_BLOCKING'] = \"1\"\nimport tensorflow as tf \ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    tpu_strategy = tf.distribute.experimental.TPUStrategy(tpu)\n    BATCH_SIZE = tpu_strategy.num_replicas_in_sync * 64\n    print(\"Running on TPU:\", tpu.master())\n    print(f\"Batch Size: {BATCH_SIZE}\")\n    \nexcept ValueError:\n    strategy = tf.distribute.get_strategy()\n    BATCH_SIZE = 512\n    print(f\"Running on {strategy.num_replicas_in_sync} replicas\")\n    print(f\"Batch Size: {BATCH_SIZE}\")\n\n\n# Utils\n# Function to get hardware strategy\ndef get_hardware_strategy():\n    try:\n        # TPU detection. No parameters necessary if TPU_NAME environment variable is\n        # set: this is always the case on Kaggle.\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        print('Running on TPU ', tpu.master())\n    except ValueError:\n        tpu = None\n\n    if tpu:\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        tf.config.optimizer.set_jit(True)\n    else:\n        # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n        strategy = tf.distribute.get_strategy()\n\n    return tpu, strategy\ntpu, strategy = get_hardware_strategy()    ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","id":"FtGcTy8sgrx1","outputId":"875550be-4174-433c-b4a4-2276da636097","execution":{"iopub.status.busy":"2022-08-09T23:18:59.588221Z","iopub.execute_input":"2022-08-09T23:18:59.588705Z","iopub.status.idle":"2022-08-09T23:19:04.665062Z","shell.execute_reply.started":"2022-08-09T23:18:59.588660Z","shell.execute_reply":"2022-08-09T23:19:04.663991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os,random,time,datetime,math\nimport cudf as cudf\nimport pandas as pd\nimport cupy as cupy  \nimport numpy as np\nfrom sklearn.preprocessing import PolynomialFeatures,LabelEncoder\nfrom sklearn.preprocessing import StandardScaler, QuantileTransformer, OneHotEncoder, PowerTransformer\nimport joblib\nimport pathlib\nfrom IPython.display import Image\nimport tqdm\nimport pickle\nfrom scipy import stats\nfrom tqdm import tqdm\nfrom tqdm.keras import TqdmCallback\nfrom sklearn.model_selection import StratifiedKFold\nfrom adabelief_tf import AdaBeliefOptimizer\nfrom colorama import Fore, Back, Style\n# tf.config.threading.set_inter_op_parallelism_threads(4)\nimport tensorflow_addons as tfa\nfrom tensorflow.keras.models import Model, load_model,model_from_json\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, LearningRateScheduler, EarlyStopping\nfrom tensorflow.keras.layers import Dense, Input, InputLayer, Add, Concatenate, Dropout, BatchNormalization, Conv1D, Reshape, Flatten, AveragePooling1D, MaxPool1D\nfrom tensorflow.keras.utils import plot_model\nfrom tensorflow.keras.losses import binary_crossentropy\nimport tensorflow.keras.backend as K\nfrom tensorflow.keras.layers import Input, Dense, LSTM,GRU,  Conv1D,Dropout,Bidirectional,Multiply\n\nimport seaborn as sns \nimport matplotlib.pyplot as plt\nplt.style.use('ggplot')\nplt.rcParams['figure.figsize'] = (16,9)\nplt.rcParams[\"figure.facecolor\"] = '#FFFACD'\nplt.rcParams[\"axes.facecolor\"] = '#FFFFE0'\nplt.rcParams[\"axes.grid\"] = True \nplt.rcParams[\"grid.alpha\"] = 0.5\nplt.rcParams[\"grid.linestyle\"] = '--'\n\n\nclass CFG:\n    seed = 42\n    INPUT = \"../input\"\n    TRAIN = True \n    INFER = False\n    n_folds = 5\n    target ='target'\n    DEBUG= False \n    ADD_CAT = False\n    ADD_LAG = False \n    TRIM = True \n    max_epochs = 100\n    batch_size = 2*1024  \n    train_folds = [0]\n    model_dir = f\"..input/amex-mlp/dnn1\"\n\npath = f'{CFG.INPUT}/amex-data-integer-dtypes-parquet-format'   \ndef seed_everything(seed):\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n\nseed_everything(CFG.seed)    ","metadata":{"id":"nD6axCLvgrx3","execution":{"iopub.status.busy":"2022-08-09T23:19:04.667923Z","iopub.execute_input":"2022-08-09T23:19:04.668627Z","iopub.status.idle":"2022-08-09T23:19:07.545925Z","shell.execute_reply.started":"2022-08-09T23:19:04.668586Z","shell.execute_reply":"2022-08-09T23:19:07.544768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature Engineering 📱📶","metadata":{"id":"SWPQupqBgrx4"}},{"cell_type":"code","source":"\nCID =\"customer_ID\"\nTIME = \"S_2\"\nTARGET = \"target\"","metadata":{"id":"jX87V9Qrrc1s","execution":{"iopub.status.busy":"2022-08-09T23:19:07.552381Z","iopub.execute_input":"2022-08-09T23:19:07.555027Z","iopub.status.idle":"2022-08-09T23:19:07.562476Z","shell.execute_reply.started":"2022-08-09T23:19:07.554984Z","shell.execute_reply":"2022-08-09T23:19:07.561453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Main FE","metadata":{"id":"f8ZAI9DwVbD9"}},{"cell_type":"code","source":"def get_not_used():  \n  return ['row_id', 'customer_ID', 'target', 'cid', 'S_2',\"D_103\",\"D_139\"]\n\nstats = ['mean', 'min', 'max']\nfeatures_avg = ['S_2_wk','B_1', 'B_2', 'B_3', 'B_4', 'B_5', 'B_6', 'B_8', 'B_9', 'B_10', 'B_11', 'B_12', 'B_13', 'B_14', 'B_15', 'B_16', 'B_17', 'B_18',\n                'B_19', 'B_20', 'B_21', 'B_22', 'B_23', 'B_24', 'B_25', 'B_28', 'B_29', 'B_30', 'B_32', 'B_33', 'B_37', 'B_38', 'B_39', 'B_40', 'B_41', 'B_42',\n                'D_39', 'D_41', 'D_42', 'D_43', 'D_44', 'D_45', 'D_46', 'D_47', 'D_48', 'D_50', 'D_51', 'D_53', 'D_54', 'D_55', 'D_58', 'D_59', 'D_60', 'D_61', \n                'D_62', 'D_65', 'D_66', 'D_69', 'D_70', 'D_71', 'D_72', 'D_73', 'D_74', 'D_75', 'D_76', 'D_77', 'D_78', 'D_80', 'D_82', 'D_84', 'D_86', 'D_91', \n                'D_92', 'D_94', 'D_96', 'D_103', 'D_104', 'D_108', 'D_112', 'D_113', 'D_114', 'D_115', 'D_117', 'D_118', 'D_119', 'D_120', 'D_121', 'D_122', 'D_123',\n                'D_124', 'D_125', 'D_126', 'D_128', 'D_129', 'D_131', 'D_132', 'D_133', 'D_134', 'D_135', 'D_136', 'D_140', 'D_141', 'D_142', 'D_144', 'D_145',\n                'P_2', 'P_3', 'P_4', 'R_1', 'R_2', 'R_3', 'R_7', 'R_8', 'R_9', 'R_10', 'R_11', 'R_14', 'R_15', 'R_16', 'R_17', 'R_20', 'R_21', 'R_22', 'R_24', \n                'R_26', 'R_27', 'S_3', 'S_5', 'S_6', 'S_7', 'S_9', 'S_11', 'S_12', 'S_13', 'S_15', 'S_16', 'S_18', 'S_22', 'S_23', 'S_25', 'S_26']\nfeatures_min = ['B_2', 'B_4', 'B_5', 'B_9', 'B_13', 'B_14', 'B_15', 'B_16', 'B_17', 'B_19', 'B_20', 'B_28', 'B_29', 'B_33', 'B_36', 'B_42', 'D_39',\n                'D_41', 'D_42', 'D_45', 'D_46', 'D_48', 'D_50', 'D_51', 'D_53', 'D_55', 'D_56', 'D_58', 'D_59', 'D_60', 'D_62', 'D_70', 'D_71', 'D_74', \n                'D_75', 'D_78', 'D_83', 'D_102', 'D_112', 'D_113', 'D_115', 'D_118', 'D_119', 'D_121', 'D_122', 'D_128', 'D_132', 'D_140', 'D_141', 'D_144',\n                'D_145', 'P_2', 'P_3', 'R_1', 'R_27', 'S_3', 'S_5', 'S_7', 'S_9', 'S_11', 'S_12', 'S_23', 'S_25']\nfeatures_std = [\"B_4\",\"R_1\",\"B_38\",\"B_40\",\"B_3\",\"D_39\",\"D_44\",\"S_25\",\"D_75\",\"D_43\",\"B_23\",\"P_2\",\"D_70\",\"S_26\",\"R_3\",\"B_19\",\"D_65\",\n                \"S_15\",\"B_28\",\"B_30\",\"B_6\",\"B_9\",\"S_12\",\"D_55\",\"S_23\",\"D_48\",\"D_61\",\"D_133\",\"B_17\",\"D_144\",\"D_42\",\"D_74\",\n                \"S_11\",\"S_7\",\"S_13\",\"D_58\",\"D_104\",\"S_3\",\"P_3\",\"D_121\",\"B_13\",\"B_20\",\"S_22\",\"B_22\",\"B_16\",\"B_5\",\"D_45\",\"D_62\",\n                \"B_12\",\"B_24\",\"D_117\",\"D_47\",\"B_8\",\"S_5\",\"D_60\",\"D_59\",\"P_4\",\"B_11\",\"D_119\",\"D_115\",\"D_113\",\"D_46\",\n                \"R_16\",\"S_2_wk\",\"B_2\",\"B_1\",\"R_27\",\"B_14\",\"D_78\",\"B_18\",\"D_124\",\"D_126\",\"B_21\",\"D_142\",\"D_131\",\"D_136\",\"D_71\",\n                \"B_37\",\"D_53\",\"S_9\",\"D_112\",\"D_118\",\"D_77\",\"B_15\",\"R_10\",\"D_80\",\"R_11\",\n                \"D_120\",\"S_16\",\"D_128\",\"B_33\",\"S_6\",\"B_10\",\"D_122\",\"D_132\",\"D_69\",\"B_25\",\"D_41\",\"D_141\"]                 \nfeatures_max = ['B_1', 'B_2', 'B_3', 'B_4', 'B_5', 'B_6', 'B_7', 'B_8', 'B_9', 'B_10', 'B_12', 'B_13', 'B_14', 'B_15', 'B_16', 'B_17', 'B_18', 'B_19',\n                'B_21', 'B_23', 'B_24', 'B_25', 'B_29', 'B_30', 'B_33', 'B_37', 'B_38', 'B_39', 'B_40', 'B_42', 'D_39', 'D_41', 'D_42', 'D_43', 'D_44',\n                'D_45', 'D_46', 'D_47', 'D_48', 'D_49', 'D_50', 'D_52', 'D_55', 'D_56', 'D_58', 'D_59', 'D_60', 'D_61', 'D_63', 'D_64', 'D_65', 'D_70',\n                'D_71', 'D_72', 'D_73', 'D_74', 'D_76', 'D_77', 'D_78', 'D_80', 'D_82', 'D_84', 'D_91', 'D_102', 'D_105', 'D_107', 'D_110', 'D_111', 'D_112',\n                'D_115', 'D_116', 'D_117', 'D_118', 'D_119', 'D_121', 'D_122', 'D_123', 'D_124', 'D_125', 'D_126', 'D_128', 'D_131', 'D_132', 'D_133',\n                'D_134', 'D_135', 'D_136', 'D_138', 'D_140', 'D_141', 'D_142', 'D_144', 'D_145', 'P_2', 'P_3', 'P_4', 'R_1', 'R_3', 'R_5', 'R_6', 'R_7',\n                'R_8', 'R_10', 'R_11', 'R_14', 'R_17', 'R_20', 'R_26', 'R_27', 'S_3', 'S_5', 'S_7', 'S_8', 'S_11', 'S_12', 'S_13', 'S_15', 'S_16', 'S_22',\n                'S_23', 'S_24', 'S_25', 'S_26', 'S_27']\nfeatures_last = ['B_1', 'B_2', 'B_3', 'B_4', 'B_5', 'B_6', 'B_7', 'B_8', 'B_9', 'B_10', 'B_11', 'B_12', 'B_13', 'B_14', 'B_15', 'B_16', 'B_17', \n                 'B_18', 'B_19', 'B_20', 'B_21', 'B_22', 'B_23', 'B_24', 'B_25', 'B_26', 'B_28', 'B_29', 'B_30', 'B_32', 'B_33', 'B_36', 'B_37', 'B_38',\n                 'B_39', 'B_40', 'B_41', 'B_42', 'D_39', 'D_41', 'D_42', 'D_43', 'D_44', 'D_45', 'D_46', 'D_47', 'D_48', 'D_49', 'D_50', 'D_51', 'D_52',\n                 'D_53', 'D_54', 'D_55', 'D_56', 'D_58', 'D_59', 'D_60', 'D_61', 'D_62', 'D_63', 'D_64', 'D_65', 'D_69', 'D_70', 'D_71', 'D_72', 'D_73', 'D_75', 'D_76', 'D_77', 'D_78', 'D_79', 'D_80', 'D_81', 'D_82', 'D_83', 'D_86', 'D_91', 'D_96', 'D_105', 'D_106', 'D_112', 'D_114', 'D_119', 'D_120', 'D_121', 'D_122', 'D_124', 'D_125', 'D_126', 'D_127', 'D_130', 'D_131', 'D_132', 'D_133', 'D_134', 'D_138', 'D_140', 'D_141', 'D_142', 'D_145', 'P_2', 'P_3', 'P_4', 'R_1', 'R_2', 'R_3', 'R_4', 'R_5', 'R_6', 'R_7', 'R_8', 'R_9', 'R_10', 'R_11', 'R_12', 'R_13', 'R_14', 'R_15', 'R_19', 'R_20', 'R_26', 'R_27', 'S_3',\n                 'S_5', 'S_6', 'S_7', 'S_8', 'S_9', 'S_11', 'S_12', 'S_13', 'S_16', 'S_19', 'S_20', 'S_22', 'S_23', 'S_24', 'S_25', 'S_26', 'S_27']\n# Feature Engineering on credit risk\nspend_p=[ 'S_3',  'S_5', 'S_6', 'S_7', 'S_8', 'S_9', 'S_11', 'S_12', 'S_13', 'S_15', 'S_16', 'S_17', 'S_18', 'S_19', 'S_20', 'S_22', 'S_23', 'S_24', 'S_25', 'S_26', 'S_27']\nbalance_p = ['B_1', 'B_2', 'B_3',  'B_5', 'B_6', 'B_7', 'B_8', 'B_9', 'B_10', 'B_11', 'B_12', 'B_13', 'B_14', 'B_15',  'B_17', 'B_18',  'B_21',   'B_23', 'B_24', 'B_25', 'B_26', 'B_27', 'B_28',  'B_36', 'B_37',  'B_40',    ]\npayment_p = ['P_2', 'P_3', 'P_4']\ncat_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120',\n            'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\ndelq = ['D_39',\n                'D_41', 'D_42', 'D_45', 'D_46', 'D_48', 'D_50', 'D_51', 'D_53', 'D_55', 'D_56', 'D_58', 'D_59', 'D_60', 'D_62', 'D_70', 'D_71', 'D_74', \n                'D_75', 'D_78', 'D_83', 'D_102', 'D_112', 'D_113', 'D_115', 'D_118', 'D_119', 'D_121', 'D_122', 'D_128', 'D_132', 'D_140', 'D_141', 'D_144',\n                'D_145']                \ncat_cols_avg = [col for col in cat_cols if col in features_avg]\nnot_used = get_not_used()            \ng_num_cols = []\ndef preprocess(df):\n    df['row_id'] = cupy.arange(df.shape[0])\n    not_used = get_not_used()\n    # Drop cols https://www.kaggle.com/code/raddar/redundant-features-amex/notebook\n    df=df.drop([\"D_103\",\"D_139\"],axis=1)\n    num_cols = [col for col in df.columns if col not in cat_cols+not_used]   \n    globals()['g_num_cols'] = num_cols\n    for col in df.columns:\n        if col not in not_used+cat_cols:\n           df[col] = df[col].round(decimals=2)\n    print(f\"Starting fe [{len(df.columns)}]\") \n    dgs=add_stats_step(df, num_cols)\n    # END Custom cherry picked Stats\n    print(f\"Stats added and calculated [{len(df.columns)}]\")    \n\n    # Add s2 count as a feature ( Number of spends)\n    s2_count = df.groupby(\"customer_ID\")['S_2'].agg(['count']) \n    s2_count.columns = ['S_2_Count']\n    s2_count.reset_index(inplace = True)     \n    dgs.append(s2_count)\n    print(f\"Stats added and calculated [{len(s2_count.columns)}]\")    \n    del s2_count; gc.collect() \n\n    # Add Lag Columns \n    if CFG.ADD_LAG:\n      train_num_agg = df.groupby(\"customer_ID\")[num_cols].agg(['first', 'last'])#payment_p+balance_p+spend_p\n      train_num_agg.columns = ['_'.join(x) for x in train_num_agg.columns]\n      train_num_agg.reset_index(inplace = True) \n      for col in train_num_agg:\n        if 'last' in col and col.replace('last', 'first') in train_num_agg:\n                    train_num_agg[col + '_lag_sub'] = train_num_agg[col] - train_num_agg[col.replace('last', 'first')]\n                    #train_num_agg[col + '_lag_div'] = train_num_agg[col] / train_num_agg[col.replace('last', 'first')]            \n      train_num_agg.drop([col for col in train_num_agg.columns if \"last\" in col],axis=1, inplace=True)\n      dgs.append(train_num_agg)\n      del train_num_agg        \n    \n    # compute \"after pay\" features\n    for bcol in [f'B_{i}' for i in [11,14,17]]+['D_39','D_131']+[f'S_{i}' for i in [16,23]]:\n        for pcol in ['P_2','P_3']:\n            if bcol in df.columns:\n                df[f'{bcol}-{pcol}'] = df[bcol] - df[pcol]\n    #\n    df['S_2'] = cudf.to_datetime(df['S_2'])\n    df['cid'], _ = df.customer_ID.factorize()    \n\n    # Add sundays count as a feature \n    s2_count = df[df.S_2.dt.dayofweek == 6].groupby(\"customer_ID\")['S_2'].agg(['count']) \n    s2_count.columns = ['S_2_Sun_Count']\n    s2_count.reset_index(inplace = True)     \n    dgs.append(s2_count)\n    print(f\"sundays count added and calculated [{len(s2_count.columns)}]\")   \n    del s2_count; gc.collect()     \n    if CFG.ADD_CAT:\n      train_cat_agg = df.groupby(\"customer_ID\")[cat_cols].agg(['count', 'nunique']) \n      train_cat_agg.columns = ['_'.join(x) for x in train_cat_agg.columns]\n      train_cat_agg.reset_index(inplace = True)     \n      dgs.append(train_cat_agg)\n      del train_cat_agg; gc.collect() \n      train_cat_mean = df.groupby(\"customer_ID\")[cat_cols_avg].agg(['mean']) \n      train_cat_mean.columns = ['_'.join(x) for x in train_cat_mean.columns]\n      train_cat_mean.reset_index(inplace = True)    \n      print(f\"Added cat mean cols [{train_cat_mean.columns}]\")   \n      dgs.append(train_cat_mean)\n      del train_cat_mean; gc.collect() \n      print(f\"CAT features added {len(df.columns)}\") \n    # restore the original row order by sorting row_id\n    df = df.sort_values('row_id')\n    df = df.drop(['row_id'],axis=1)\n    return df, dgs\n\ndef add_stats_step(df, cols):\n    n = 50\n    dgs = []\n    for i in range(0,len(cols),n):\n        s = i\n        e = min(s+n, len(cols))\n        dg = add_stats_one_shot(df, cols[s:e])\n        dgs.append(dg)\n    return dgs\n\ndef add_stats_one_shot(df, cols):\n    \n    dg = df.groupby('customer_ID').agg({col:stats for col in cols})\n    out_cols = []\n    for col in cols:\n        out_cols.extend([f'{col}_{s}' for s in stats])\n    dg.columns = out_cols\n    dg = dg.reset_index()\n    return dg\n\ndef load_test_iter(path, chunks=4):\n    \n    test_rows = 11363762\n    chunk_rows = test_rows // chunk\n    test = cudf.read_parquet(f'{path}/test.parquet',\n                             columns=['customer_ID','S_2'],\n                             num_rows=test_rows)\n    test = get_segment(test)\n    start = 0\n    while start < test.shape[0]:\n        if start+chunk_rows < test.shape[0]:\n            end = test['cus_count'].values[start+chunk_rows]\n        else:\n            end = test['cus_count'].values[-1]\n        end = int(end)\n        df = cudf.read_parquet(f'{path}/test.parquet',\n                               num_rows = end-start, skiprows=start)\n        start = end\n        yield process_data(df)\n    \n\ndef load_train(path):\n    train = cudf.read_parquet(f'{path}/train.parquet')\n    \n    train = process_data(train)\n    trainl = cudf.read_csv(f'{CFG.INPUT}/amex-default-prediction/train_labels.csv')\n    trainl.target = trainl.target.astype('int8')\n    train = train.merge(trainl, on='customer_ID', how='left')\n    return train\n\ndef process_data(df, infer = False):\n    df,dgs = preprocess(df) \n    df = df.drop_duplicates('customer_ID',keep='last')\n    for dg in dgs:\n        df = df.merge(dg, on='customer_ID', how='left')\n        # drop specific non impactful cols \n    del dgs; gc.collect()    \n    if CFG.TRIM:\n      drop_col = [col for  col in df.columns if ((\"std\" in col) and (col.replace(\"_std\",\"\") not in features_std))]\n      print(f\"Dropping {drop_col}\")\n      df=df.drop(drop_col,axis=1)      \n      drop_col = [col for  col in df.columns if ((\"min\" in col) and (col.replace(\"_min\",\"\") not in features_min))]\n      print(f\"Dropping {drop_col}\")\n      df=df.drop(drop_col,axis=1)\n      drop_col = [col for  col in df.columns if  ((\"max\" in col) and (col.replace(\"_max\",\"\") not in features_max))]\n      print(f\"Dropping {drop_col}\")\n      df=df.drop(drop_col,axis=1)\n      drop_col = [col for  col in df.columns if  ((\"mean\" in col) and (col.replace(\"_mean\",\"\") not in features_avg))]\n      print(f\"Dropping {drop_col}\")\n      df=df.drop(drop_col,axis=1)            \n    diff_cols = [col for col in df.columns if col.endswith('_diff')]\n    df = df.drop(diff_cols,axis=1)\n    print(f\"All stats merged {len(df.columns)}\")    \n    # add mean - last features and More custom features\n    df[\"P2B9\"] = df[\"P_2\"] / df[\"B_9\"] \n    math_col = globals()['g_num_cols']\n    for pcol in math_col:\n      if pcol+\"_mean\" in df.columns:\n        df[f'{pcol}-mean'] = df[pcol] - df[pcol+\"_mean\"] \n      if (pcol+\"_min\" in df.columns) and (pcol+\"_max\" in df.columns):   \n        df[f'{pcol}_lag_min'] = df[pcol+\"_min\"] - df[pcol] \n        df[f'{pcol}_lag_max'] = df[pcol+\"_max\"] - df[pcol]\n\n    print(f\"Addding col-mean {len(df.columns)} cols {math_col}\")    \n    _ = gc.collect()\n    # Impute missing values\n    nm_cols = [col for col in df.columns if (col not in cat_cols+get_not_used()) and df[col].dtype != 'int32']\n    df= df.fillna(-1) #this applies to numerical columns \n    ## Replace inf with zeros \n    df = cudf.from_pandas(df.to_pandas().replace([np.inf,-np.inf], -1))\n    # Reduce memory\n    for c in df.columns:\n        if c in get_not_used(): continue\n        if str( df[c].dtype )=='int64':\n          df[c] = df[c].astype('int32')\n        if str(df[c].dtype )=='float64':\n          df[c] = df[c].astype('float32')\n    return df\n\ndef get_segment(test):\n    dg = test.groupby('customer_ID').agg({'S_2':'count'})\n    dg.columns = ['cus_count']\n    dg = dg.reset_index()\n    dg['cid'],_ = dg['customer_ID'].factorize()\n    dg = dg.sort_values('cid')\n    dg['cus_count'] = dg['cus_count'].cumsum()\n    \n    test = test.merge(dg, on='customer_ID', how='left')\n    test = test.sort_values(['cid','S_2'])\n    assert test['cus_count'].values[-1] == test.shape[0]\n    return test","metadata":{"id":"IgC1UYiKgrx4","execution":{"iopub.status.busy":"2022-08-09T23:19:07.567250Z","iopub.execute_input":"2022-08-09T23:19:07.567914Z","iopub.status.idle":"2022-08-09T23:19:07.659449Z","shell.execute_reply.started":"2022-08-09T23:19:07.567882Z","shell.execute_reply":"2022-08-09T23:19:07.658182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model","metadata":{"id":"ksVFY-18grx6"}},{"cell_type":"code","source":"from tensorflow import keras\nfrom tensorflow.keras import layers\nNUM_TRANSFORMER_BLOCKS = 4  # Number of transformer blocks.\nNUM_HEADS = 2  # Number of attention heads.\nEMBEDDING_DIMS = 8  # Embedding dimensions of the categorical features.\nDROPOUT_RATE = 0.2\nMLP_HIDDEN_UNITS_FACTORS = [\n    2,\n    1,\n]  # MLP hidden layer units, as factors of the number of inputs.\nNUM_MLP_BLOCKS = 2  # Number of MLP blocks in the baseline model.","metadata":{"id":"KnZXa8emD748","jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-08-09T23:19:07.661194Z","iopub.execute_input":"2022-08-09T23:19:07.661569Z","iopub.status.idle":"2022-08-09T23:19:07.670863Z","shell.execute_reply.started":"2022-08-09T23:19:07.661515Z","shell.execute_reply":"2022-08-09T23:19:07.669118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def encode_inputs(inputs, embedding_dims):\n\n    encoded_categorical_feature_list = []\n    numerical_feature_list = []\n\n    for feature_name in inputs:\n        if feature_name in cat_cols:\n\n            # Get the vocabulary of the categorical feature.\n            vocabulary = CATEGORICAL_FEATURES_WITH_VOCABULARY[feature_name]\n\n            # Create a lookup to convert string values to an integer indices.\n            # Since we are not using a mask token nor expecting any out of vocabulary\n            # (oov) token, we set mask_token to None and  num_oov_indices to 0.\n            \n            \"\"\"lookup = layers.StringLookup(\n                vocabulary=vocabulary,\n                mask_token=None,\n                num_oov_indices=0,\n                output_mode=\"int\",\n            )\n\n            # Convert the string input values into integer indices.\n            encoded_feature = lookup(inputs[feature_name])\n\"\"\"         \n            encoded_feature =inputs[feature_name]\n            # Create an embedding layer with the specified dimensions.\n            embedding = layers.Embedding(\n                input_dim=len(vocabulary), output_dim=embedding_dims\n            )\n\n            # Convert the index values to embedding representations.\n            encoded_categorical_feature = embedding(encoded_feature)\n            encoded_categorical_feature_list.append(encoded_categorical_feature)\n\n        else:\n\n            # Use the numerical features as-is.\n            numerical_feature = tf.expand_dims(inputs[feature_name], -1)\n            numerical_feature_list.append(numerical_feature)\n\n    return encoded_categorical_feature_list, numerical_feature_list","metadata":{"id":"GkWr5pgCr6mK","jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-08-09T23:19:07.672810Z","iopub.execute_input":"2022-08-09T23:19:07.673261Z","iopub.status.idle":"2022-08-09T23:19:07.683526Z","shell.execute_reply.started":"2022-08-09T23:19:07.673220Z","shell.execute_reply":"2022-08-09T23:19:07.682385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model_inputs():\n    inputs = {}\n    for feature_name in FEATURE_NAMES:\n        if feature_name in NUMERIC_FEATURE_NAMES:\n            inputs[feature_name] = layers.Input(\n                name=feature_name, shape=(), dtype=tf.float32\n            )\n        else:\n            inputs[feature_name] = layers.Input(\n                name=feature_name, shape=(), dtype=tf.string\n            )\n            \n    return inputs\n\ndef create_mlp(hidden_units, dropout_rate, activation, normalization_layer, name=None):\n\n    mlp_layers = []\n    for units in hidden_units:\n        mlp_layers.append(normalization_layer),\n        mlp_layers.append(layers.Dense(units, activation=activation))\n        mlp_layers.append(layers.Dropout(dropout_rate))\n\n    return keras.Sequential(mlp_layers, name=name)\n\ndef create_tabtransformer_classifier(\n    num_transformer_blocks,\n    num_heads,\n    embedding_dims,\n    mlp_hidden_units_factors,\n    dropout_rate,\n    use_column_embedding=False,\n):\n\n    # Create model inputs.\n    inputs = Input(shape=(len(FEATURE_NAMES), ))\n    embeddings = []\n    \n    for k in range(11): \n      vocabulary = CATEGORICAL_FEATURES_WITH_VOCABULARY[cat_cols[k]]\n      #print(f\"cat {cat_cols[k]} index {k} len {len(vocabulary)}\")\n      emb = tf.keras.layers.Embedding(len(vocabulary),embedding_dims)\n      embeddings.append(emb(inputs[:,k]))\n    encoded_categorical_features = tf.stack(embeddings, axis=1)\n    numerical_features = layers.concatenate([inputs[:,11:]])\n    # Add column embedding to categorical feature embeddings.\n    if use_column_embedding:\n        num_columns = encoded_categorical_features.shape[1]\n        column_embedding = layers.Embedding(\n            input_dim=num_columns, output_dim=embedding_dims\n        )\n        column_indices = tf.range(start=0, limit=num_columns, delta=1)\n        encoded_categorical_features = encoded_categorical_features + column_embedding(\n            column_indices\n        )\n\n    # Create multiple layers of the Transformer block.\n    for block_idx in range(num_transformer_blocks):\n        # Create a multi-head attention layer.\n        attention_output = layers.MultiHeadAttention(\n            num_heads=num_heads,\n            key_dim=embedding_dims,\n            dropout=dropout_rate,\n            name=f\"multihead_attention_{block_idx}\",\n        )(encoded_categorical_features, encoded_categorical_features)\n        # Skip connection 1.\n        x = layers.Add(name=f\"skip_connection1_{block_idx}\")(\n            [attention_output, encoded_categorical_features]\n        )\n        # Layer normalization 1.\n        x = layers.LayerNormalization(name=f\"layer_norm1_{block_idx}\", epsilon=1e-6)(x)\n        # Feedforward.\n        feedforward_output = create_mlp(\n            hidden_units=[embedding_dims],\n            dropout_rate=dropout_rate,\n            activation=keras.activations.gelu,\n            normalization_layer=layers.LayerNormalization(epsilon=1e-6),\n            name=f\"feedforward_{block_idx}\",\n        )(x)\n        # Skip connection 2.\n        x = layers.Add(name=f\"skip_connection2_{block_idx}\")([feedforward_output, x])\n        # Layer normalization 2.\n        encoded_categorical_features = layers.LayerNormalization(\n            name=f\"layer_norm2_{block_idx}\", epsilon=1e-6\n        )(x)\n\n    # Flatten the \"contextualized\" embeddings of the categorical features.\n    categorical_features = layers.Flatten()(encoded_categorical_features)\n    # Apply layer normalization to the numerical features.\n    numerical_features = layers.LayerNormalization(epsilon=1e-6)(numerical_features)\n    # Prepare the input for the final MLP block.\n    features = layers.concatenate([categorical_features, numerical_features])\n\n    # Compute MLP hidden_units.\n    mlp_hidden_units = [\n        factor * features.shape[-1] for factor in mlp_hidden_units_factors\n    ]\n    # Create final MLP.\n    features = create_mlp(\n        hidden_units=mlp_hidden_units,\n        dropout_rate=dropout_rate,\n        activation=keras.activations.selu,\n        normalization_layer=layers.BatchNormalization(),\n        name=\"MLP\",\n    )(features)\n\n    # Add a sigmoid as a binary classifer.\n    outputs = layers.Dense(units=1, activation=\"sigmoid\", name=\"sigmoid\")(features)\n    \n    model = keras.Model(inputs=inputs, outputs=outputs)\n    return model","metadata":{"id":"VlQKE0UQDKQc","jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-08-09T23:19:07.685554Z","iopub.execute_input":"2022-08-09T23:19:07.685993Z","iopub.status.idle":"2022-08-09T23:19:07.712851Z","shell.execute_reply.started":"2022-08-09T23:19:07.685955Z","shell.execute_reply":"2022-08-09T23:19:07.711737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tab Transformer\n\ndef my_model():\n  tabtransformer_model = create_tabtransformer_classifier(  \n      num_transformer_blocks=NUM_TRANSFORMER_BLOCKS,\n      num_heads=NUM_HEADS,\n      embedding_dims=EMBEDDING_DIMS,\n      mlp_hidden_units_factors=MLP_HIDDEN_UNITS_FACTORS,\n      dropout_rate=DROPOUT_RATE,\n  )\n  return tabtransformer_model\n","metadata":{"id":"coOPXOi_grx7","jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-08-09T23:19:07.717448Z","iopub.execute_input":"2022-08-09T23:19:07.718311Z","iopub.status.idle":"2022-08-09T23:19:07.728345Z","shell.execute_reply.started":"2022-08-09T23:19:07.718271Z","shell.execute_reply":"2022-08-09T23:19:07.727229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Metrics","metadata":{"id":"eUS5KsVDgrx8"}},{"cell_type":"code","source":"def amex_metric(y_true, y_pred, return_components=False) -> float:\n    \"\"\"Amex metric for ndarrays\"\"\"\n    def top_four_percent_captured(df) -> float:\n        \"\"\"Corresponds to the recall for a threshold of 4 %\"\"\"\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(df) -> float:\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(df) -> float:\n        \"\"\"Corresponds to 2 * AUC - 1\"\"\"\n        df2 = pd.DataFrame({'target': df.target, 'prediction': df.target})\n        df2.sort_values('prediction', ascending=False, inplace=True)\n        return weighted_gini(df) / weighted_gini(df2)\n\n    df = pd.DataFrame({'target': y_true.ravel(), 'prediction': y_pred.ravel()})\n    df.sort_values('prediction', ascending=False, inplace=True)\n    g = normalized_weighted_gini(df)\n    d = top_four_percent_captured(df)\n\n    if return_components: return g, d, 0.5 * (g + d)\n    del df\n    gc.collect()\n    return 0.5 * (g + d)\n#Keras\ndef DiceBCELoss(targets, inputs, smooth=1e-6):    \n    BCE =  binary_crossentropy(targets, inputs)\n    intersection = K.sum(K.dot(targets, inputs))    \n    dice_loss = 1 - (2*intersection + smooth) / (K.sum(targets) + K.sum(inputs) + smooth)\n    Dice_BCE = BCE + dice_loss\n    \n    return Dice_BCE\nALPHA = 0.8\nGAMMA = 2\ndef FocalLoss(targets, inputs, alpha=ALPHA, gamma=GAMMA):    \n    BCE = K.binary_crossentropy(targets, inputs)\n    BCE_EXP = K.exp(-BCE)\n    focal_loss = K.mean(alpha * K.pow((1-BCE_EXP), gamma) * BCE)\n    \n    return focal_loss","metadata":{"id":"KiUMoni5grx8","jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-08-09T23:19:07.735993Z","iopub.execute_input":"2022-08-09T23:19:07.738688Z","iopub.status.idle":"2022-08-09T23:19:07.758691Z","shell.execute_reply.started":"2022-08-09T23:19:07.738641Z","shell.execute_reply":"2022-08-09T23:19:07.757440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load data and add feature","metadata":{"id":"e8ZNozN7grx9"}},{"cell_type":"code","source":"%%time \nimport torch, gc \ntorch.cuda.empty_cache()\n_ = gc.collect()  \nCATEGORICAL_FEATURES_WITH_VOCABULARY = {\n}\n\nif CFG.TRAIN :\n    fe = \"/content/train_fe_v1659712950.733633.pickle\"\n    if os.path.exists(fe):\n        train = cudf.read_pickle(fe)\n    else:  \n        path = f'{CFG.INPUT}/amex-data-integer-dtypes-parquet-format'\n        train = load_train(path)       \n        print(\"Saving FE to file\")\n        #train.to_pickle(f\"train_fe_v{time.time()}.pickle\")\n        gc.collect()\n    drop_col = [col for col in train.columns if ((\"_diff\" in col) or (\"_TE11\" in col) or (\"_TE10\" in col) or (\"_TE9\" in col) or (\"_TE8\" in col) or (\"_TE7\" in col)) ]\n    print(f\"Dropping zero importance {drop_col}\")\n    train = train.drop(drop_col,axis=1) .reset_index(drop=True)      \n    print(train.shape)  \n    features = [col for col in train.columns if col not in  get_not_used()]\n    FEATURE_NAMES = [col for col in train.columns if col not in  get_not_used()]\n    NUMERIC_FEATURE_NAMES = [col for col in FEATURE_NAMES if col not in cat_cols]\n    print(f\"Col features {len(cat_cols)}\")\n    gc.collect()","metadata":{"id":"W3fHX37Igrx-","outputId":"2b02e1ab-79fa-4713-c672-738c6e2a720c","execution":{"iopub.status.busy":"2022-08-09T23:20:46.918594Z","iopub.execute_input":"2022-08-09T23:20:46.919744Z","iopub.status.idle":"2022-08-09T23:21:21.015706Z","shell.execute_reply.started":"2022-08-09T23:20:46.919689Z","shell.execute_reply":"2022-08-09T23:21:21.013725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if CFG.DEBUG:\n    train = train.sample(n=2000, random_state=42).reset_index(drop=True) \n    features = [col for col in train.columns if col not in  get_not_used()]\nif CFG.INFER:\n    test = process_data(cudf.read_parquet(f'/content/amex-data-integer-dtypes-parquet-format/test.parquet'),True)\n    print(f\"Test size {test.shape}\")\n\n    drop_col = [col for col in train.columns if ((\"_diff\" in col) or (\"_TE11\" in col) or (\"_TE10\" in col) or (\"_TE9\" in col) or (\"_TE8\" in col) or (\"_TE7\" in col)) ]\n    print(f\"Dropping zero importance {drop_col}\")\n    test = test.drop(drop_col,axis=1)\n    features = [col for col in test.columns if col not in  get_not_used()]    \n    FEATURE_NAMES = [col for col in test.columns if col not in  get_not_used()]\n    NUMERIC_FEATURE_NAMES = [col for col in FEATURE_NAMES if col not in cat_cols]\n    print(f\"Col features {len(cat_cols)}\")\n    gc.collect()","metadata":{"id":"7YVypOzegrx-","jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-08-09T23:20:00.596234Z","iopub.execute_input":"2022-08-09T23:20:00.597814Z","iopub.status.idle":"2022-08-09T23:20:00.606744Z","shell.execute_reply.started":"2022-08-09T23:20:00.597773Z","shell.execute_reply":"2022-08-09T23:20:00.605598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train 📖📓","metadata":{"id":"0y1TLKFDgrx_"}},{"cell_type":"code","source":"VERBOSE = 0\nCYCLES = 1\nEPOCHS = CFG.max_epochs\nUSE_PLATEAU = False\nBATCH_SIZE = CFG.batch_size\nWEIGHT_DECAY = 0.001\nLR_START = 0.01\nEPOCHS_EXPONENTIALDECAY = 100\nLR_END = 1e-5 # learning rate at the end of training\noof_predictions = np.zeros(len(train))\nscore_list = []\nhistory_list = []\n_ = gc.collect()\n\ndef fit_model(X_tr, X_va, y_tr, y_va=None, fold=0):\n    start_time = datetime.datetime.now()\n    # Scaling Helps but not running to save memory \n    le_dict = {col: LabelEncoder() for col in cat_cols }\n    for col in cat_cols:\n       X_tr[col] = le_dict[col].fit_transform(X_tr[col]) \n       X_va[col] = le_dict[col].fit_transform(X_va[col])\n    # Persist for test \n    with open(f\"le_{fold}.pickle\", 'wb') as f: pickle.dump(le_dict, f)\n    del le_dict;gc.collect()\n    # Embeddings \n    for col in cat_cols:\n        CATEGORICAL_FEATURES_WITH_VOCABULARY[col]= sorted(list(X_tr[col].unique()))\n    print(f\"Cat vocab : {CATEGORICAL_FEATURES_WITH_VOCABULARY}\")\n    # Rearrange cols\n    COLS = cat_cols + [c for c in X_tr.columns if c not in cat_cols]\n    X_tr = X_tr[COLS]\n    X_va = X_va[COLS]\n    # Use Numpy vals\n    X_va = X_va.values\n    X_tr =X_tr.values  \n    es = EarlyStopping(monitor=\"val_loss\",\n                       patience=24, \n                       verbose=VERBOSE,\n                       mode=\"min\", \n                       restore_best_weights=True) \n    if USE_PLATEAU and X_va is not None: # use early stopping\n        epochs = EPOCHS\n        lr = ReduceLROnPlateau(monitor=\"val_loss\", factor=0.7, \n                               patience=4, verbose=VERBOSE)\n        es = EarlyStopping(monitor=\"val_loss\",\n                           patience=12, \n                           verbose=VERBOSE,\n                           mode=\"min\", \n                           restore_best_weights=True)\n        callbacks = [lr, es, tf.keras.callbacks.TerminateOnNaN(),TqdmCallback(verbose=1)]\n    else: # use exponential learning rate decay rather than early stopping\n        print(\"using Exp decay\")\n        epochs = EPOCHS_EXPONENTIALDECAY\n\n        def exponential_decay(epoch):\n            # v decays from e^a to 1 in every cycle\n            # w decays from 1 to 0 in every cycle\n            # epoch == 0                  -> w = 1 (first epoch of cycle)\n            # epoch == epochs_per_cycle-1 -> w = 0 (last epoch of cycle)\n            # higher a -> decay starts with a steeper decline\n            a = 3\n            epochs_per_cycle = epochs // CYCLES\n            epoch_in_cycle = epoch % epochs_per_cycle\n            if epochs_per_cycle > 1:\n                v = math.exp(a * (1 - epoch_in_cycle / (epochs_per_cycle-1)))\n                w = (v - 1) / (math.exp(a) - 1)\n            else:\n                w = 1\n            return w * LR_START + (1 - w) * LR_END\n\n        lr = LearningRateScheduler(exponential_decay, verbose=0)\n        callbacks = [lr, tf.keras.callbacks.TerminateOnNaN(),TqdmCallback(verbose=1)]\n    checkpoint_filepath = f\"{CFG.model_dir}/model_fold{fold}_seed{CFG.seed}.h5\"   \n    if os.path.exists(checkpoint_filepath):\n        model = tf.keras.models.load_model(checkpoint_filepath) \n    else:\n      model = my_model()\n      if fold == 0:\n        print(\"Total model weights:\", model.count_params())\n        keras.utils.plot_model(model, show_shapes=True, rankdir=\"LR\")\n        Image(filename='model.png') \n      #optimizer = tfa.optimizers.AdamW(learning_rate=LR_START, weight_decay=WEIGHT_DECAY)\n      optimizer =tf.keras.optimizers.Adam(learning_rate=LR_START)\n      model.compile(optimizer=optimizer,\n                loss=tf.keras.losses.BinaryCrossentropy())\n      gc.collect() \n      history = model.fit(X_tr, y_tr, \n              validation_data=(X_va, y_va),\n              epochs=EPOCHS,\n              verbose=VERBOSE,\n              batch_size=BATCH_SIZE,\n              shuffle=True,\n              callbacks=callbacks)\n      history_list.append(history.history)\n      del X_tr, y_tr, callbacks,history, es, lr; gc.collect()\n    oof_pred = model.predict(X_va).reshape( (len(X_va), )) \n    score = amex_metric(y_va, oof_pred)\n    if os.path.exists(checkpoint_filepath):\n      print(f\"{Fore.GREEN}{Style.BRIGHT}Fold {fold} | {str(datetime.datetime.now() - start_time)[-12:-7]}\" \n          f\" |  Score: {score:.5f}{Style.RESET_ALL}\")\n    else:  \n      lastloss = f\"Training loss: {history_list[-1]['loss'][-1]:.4f} | Val loss: {history_list[-1]['val_loss'][-1]:.4f}\"\n      print(f\"{Fore.GREEN}{Style.BRIGHT}Fold {fold} | {str(datetime.datetime.now() - start_time)[-12:-7]}\"\n        f\" | {len(history_list[-1]['loss']):3} ep\"\n        f\" | {lastloss} | Score: {score:.5f}{Style.RESET_ALL}\")\n      \n    score_list.append(score)\n    oof_predictions[idx_va]+=oof_pred\n    model.save(f\"model_fold{fold}_seed{CFG.seed}.h5\") \n\ndef fit_train_models(X_tr, X_va, y_tr, y_va=None, fold=0):\n    print(f'{Fore.GREEN}{Style.BRIGHT}Training KFOLD {fold} WITH SEED {CFG.seed} {Style.RESET_ALL}')\n    oof_pred = fit_model(X_tr, X_va, y_tr, y_va, fold) \n    gc.collect()\n\n# TRAIN\ntf.keras.backend.clear_session()\nif CFG.TRAIN:\n    num_cols = [col for col in train.columns if col not in cat_cols+not_used] \n    kf = StratifiedKFold(n_splits=CFG.n_folds, shuffle= True, random_state= CFG.seed)\n    print(train.shape)\n    train = train.to_pandas() \n    torch.cuda.empty_cache()\n    _ = gc.collect()\n    \n    #scaler = StandardScaler() \n    #train   = scaler.fit_transform(train)\n    # Persist For Test\n    #with open(f\"scaler.pickle\", 'wb') as f: pickle.dump(scaler, f)\n    #del scaler;gc.collect()\n    for fold, (idx_tr, idx_va) in enumerate(kf.split(train , train.target)):\n        if fold not in CFG.train_folds:\n            continue\n        fit_train_models(train.iloc[idx_tr][features],train.iloc[idx_va][features],train.target.iloc[idx_tr], train.target.iloc[idx_va], fold)\n    print(f\"{Fore.GREEN}{Style.BRIGHT}OOF Score: {np.mean(score_list):.5f}{Style.RESET_ALL}\")\n    # SAVE OOF\n    oof_df = pd.DataFrame({'customer_ID': train['customer_ID'], 'target': train[CFG.target], 'prediction': oof_predictions})\n    oof_df.to_csv(f'mlp_{CFG.n_folds}fold_seed{CFG.seed}.csv', index = False)    ","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:21:21.019286Z","iopub.execute_input":"2022-08-09T23:21:21.019796Z","iopub.status.idle":"2022-08-09T23:39:28.431423Z","shell.execute_reply.started":"2022-08-09T23:21:21.019749Z","shell.execute_reply":"2022-08-09T23:39:28.430231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Arch ✏️📐","metadata":{}},{"cell_type":"code","source":"Image(filename='model.png') ","metadata":{"execution":{"iopub.status.busy":"2022-08-10T00:02:59.953216Z","iopub.execute_input":"2022-08-10T00:02:59.953849Z","iopub.status.idle":"2022-08-10T00:02:59.986723Z","shell.execute_reply.started":"2022-08-10T00:02:59.953806Z","shell.execute_reply":"2022-08-10T00:02:59.985855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### INFER","metadata":{"id":"5voOzOhLXUHK"}},{"cell_type":"code","source":"if CFG.INFER:\n    test_predictions = np.zeros(len(test))\n    not_used = [i for i in not_used if i in test.columns]  \n    for fold in range(CFG.n_folds):\n        scalar_path = f\"{CFG.model_dir}/le_{fold}.pickle\" \n        le_dict = pickle.load(open(scalar_path,\"rb\"))\n        for col in cat_cols:\n           test[col] = le_dict[col].transform(test[col])  \n        # Rearrange cols\n        COLS = cat_cols + [c for c in test.columns if c not in cat_cols]\n        _test = test[COLS] \n        # Use Numpy vals\n        _test = _test.values\n        checkpoint_filepath = f\"{CFG.model_dir}/model_fold{fold}_seed{CFG.seed}.h5\" \n        model = tf.keras.models.load_model(checkpoint_filepath) \n        preds = model.predict(_test).reshape( (len(_test), )) \n        test_predictions += preds / CFG.n_folds \n    test_df = pd.DataFrame({'customer_ID': test['customer_ID'], 'prediction': test_predictions})\n    test_df.to_csv(f'test_lgbm_{CFG.n_folds}fold_seed{CFG.seed}.csv', index = False) ","metadata":{"execution":{"iopub.status.busy":"2022-08-10T00:03:06.072489Z","iopub.execute_input":"2022-08-10T00:03:06.073241Z","iopub.status.idle":"2022-08-10T00:03:06.083441Z","shell.execute_reply.started":"2022-08-10T00:03:06.073200Z","shell.execute_reply":"2022-08-10T00:03:06.082215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# That's All Folks ","metadata":{}}]}