{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os, copy, gc\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\nfrom torch.utils.data import TensorDataset, DataLoader,Dataset\nfrom torch.autograd import Variable\nfrom torch.optim.lr_scheduler import MultiStepLR\nfrom sklearn.model_selection import StratifiedKFold, KFold\nfrom sklearn.metrics import accuracy_score\nfrom tqdm import tqdm\nfrom sklearn.metrics import roc_auc_score, average_precision_score\nimport matplotlib.pyplot as plt\nimport random\nprint('Using PyTorch version',torch.__version__)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:40:26.185982Z","iopub.execute_input":"2022-08-12T02:40:26.187232Z","iopub.status.idle":"2022-08-12T02:40:28.509229Z","shell.execute_reply.started":"2022-08-12T02:40:26.187107Z","shell.execute_reply":"2022-08-12T02:40:28.508159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed=42):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n\nseed_everything(seed=45)\n\ndevice = ('cuda' if torch.cuda.is_available() else 'cpu')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:40:28.511237Z","iopub.execute_input":"2022-08-12T02:40:28.511765Z","iopub.status.idle":"2022-08-12T02:40:28.583124Z","shell.execute_reply.started":"2022-08-12T02:40:28.511729Z","shell.execute_reply":"2022-08-12T02:40:28.582112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class AMEXLoader:\n\n    def __init__(self, X_3D, X_2D, y, lag, shuffle=True, batch_size=1024):\n        self.X_3D = X_3D\n        self.X_2D = X_2D\n        self.y = y\n\n        self.shuffle = shuffle\n        self.batch_size = batch_size\n        self.n_conts = self.X_3D.shape[1]\n        self.len = self.X_3D.shape[0]\n        n_batches, remainder = divmod(self.len, self.batch_size)\n\n        if remainder > 0:\n            n_batches += 1\n        self.n_batches = n_batches\n        self.remainder = remainder  # for debugging\n\n        self.idxes = np.array([i for i in range(self.len)])\n\n    def __iter__(self):\n        self.i = 0\n        if self.shuffle:\n            ridxes = self.idxes\n            np.random.shuffle(ridxes)\n            self.X_3D = self.X_3D[ridxes]\n            self.X_2D = self.X_2D[ridxes]\n            if self.y is not None:\n                self.y = self.y[ridxes]\n\n        return self\n\n    def __next__(self):\n        if self.i >= self.len:\n            raise StopIteration\n        \n        X_3D = torch.FloatTensor(self.X_3D[self.i:self.i + self.batch_size, lag:, :])\n        X_2D = torch.FloatTensor(self.X_2D[self.i:self.i + self.batch_size, :])\n        #idx = np.random.randint(self.len, size=xcont1.shape[0])\n        if self.y is not None:\n            y1 = self.y[self.i:self.i + self.batch_size]\n            #y2 = self.y[idx]\n            #y = torch.FloatTensor(np.where(y1==y2, 1, 0).astype(np.float32))\n            y1 = torch.FloatTensor(y1.astype(np.float32))\n\n        else:\n            y1 = None\n            #y = None\n            \n        #xcont2 = torch.FloatTensor(self.X_cont[idx, :, :])\n        \n\n        batch = (X_3D, X_2D, y1)#, xcont2, y)\n        self.i += self.batch_size\n        return batch\n\n    def __len__(self):\n        return self.n_batches","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:40:28.584728Z","iopub.execute_input":"2022-08-12T02:40:28.585080Z","iopub.status.idle":"2022-08-12T02:40:28.597342Z","shell.execute_reply.started":"2022-08-12T02:40:28.585045Z","shell.execute_reply":"2022-08-12T02:40:28.596334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# COMPETITION METRIC FROM Konstantin Yakovlev\n# https://www.kaggle.com/kyakovlev\n# https://www.kaggle.com/competitions/amex-default-prediction/discussion/327534\ndef amex_metric(y_true, y_pred):\n\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n    print(\"G: {:.6f}, D: {:.6f}, ALL: {:6f}\".format(gini[1]/gini[0], top_four, 0.5*(gini[1]/gini[0] + top_four)))\n    return 0.5 * (gini[1]/gini[0] + top_four)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:40:28.599824Z","iopub.execute_input":"2022-08-12T02:40:28.600926Z","iopub.status.idle":"2022-08-12T02:40:28.613049Z","shell.execute_reply.started":"2022-08-12T02:40:28.600891Z","shell.execute_reply":"2022-08-12T02:40:28.612048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def init_weights(m):\n    if type(m) == nn.Linear:\n        torch.nn.init.xavier_uniform(m.weight)\n        m.bias.data.fill_(0.01)\n\n####### LSTM model for 3D input ############################\nclass LSTM_AMEX(nn.Module):\n\n    def __init__(self, input_size, ffnn_input, hidden_size, \n                 num_layers, seq_length, activation = nn.GELU(), device=device):\n        \n        super(LSTM_AMEX, self).__init__()\n\n        self.num_layers = num_layers\n        self.input_size = input_size\n        self.hidden_size = hidden_size\n        self.seq_length = seq_length\n        self.device = device\n\n\n        self.lstm = nn.LSTM(input_size=input_size, hidden_size=hidden_size,\n                            num_layers=num_layers, batch_first=True)\n\n        \n        inp_dim = hidden_size\n        \n        self.ffnn = nn.Sequential(nn.Linear(ffnn_input, 512), \n                                        nn.Dropout(0.10),\n                                        activation,\n                                        nn.Linear(512, 256),\n                                        nn.Dropout(0.10),\n                                        activation,\n                                        nn.Linear(256, 128)\n                                        )\n        \n        \n        self.classifier = nn.Sequential(nn.Linear(inp_dim+128+input_size, 128), \n                                        nn.Dropout(0.20),\n                                        activation,\n                                        nn.Linear(128, 64),\n                                        #nn.Dropout(0.10),\n                                        activation,\n                                        nn.Linear(64, 1), \n                                        nn.Sigmoid()\n                                        )\n\n        \n        self.attention1 = nn.Sequential(\n                            nn.Linear(inp_dim, 256),\n                            nn.Tanh(),\n                            nn.Linear(256, 1),\n                            nn.Softmax(dim=1)\n                        )\n        \n\n    def forward(self, inp3D, inp2D):#, cont_x2):\n        \n        inp3D = inp3D.to(self.device)\n        inp2D = inp2D.to(self.device)\n      \n        h_0 = Variable(torch.zeros(\n            self.num_layers, inp3D.size(0), self.hidden_size)).to(device)\n\n        c_0 = Variable(torch.zeros(\n            self.num_layers, inp3D.size(0), self.hidden_size)).to(device)\n\n\n        # Propagate input through LSTM\n        x, _ = self.lstm(inp3D, (h_0, c_0))\n        weights1 = self.attention1(x)\n        x = torch.sum(weights1 * x, dim=1)\n        \n        x2 = self.ffnn(inp2D)\n        x = torch.cat([x, x2, inp3D[:, -1, :]], dim=1)\n        \n        out = self.classifier(x)\n        \n        return out","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:40:28.614508Z","iopub.execute_input":"2022-08-12T02:40:28.614797Z","iopub.status.idle":"2022-08-12T02:40:28.632162Z","shell.execute_reply.started":"2022-08-12T02:40:28.614773Z","shell.execute_reply":"2022-08-12T02:40:28.631213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"####### Custom Model ############################\n\n# Fully connected neural network with one hidden layer\nclass MLP_MODEL(nn.Module):\n    def __init__(self, input_size, hidden_size,\n                 num_layers, num_output, activation, attention=False):\n        super(MLP_MODEL, self).__init__()\n        self.meta_model = torch.nn.ModuleList()\n        for i in range(num_layers):\n            if i==0:\n                self.meta_model.append(nn.Linear(input_size, hidden_size))\n                #self.meta_model.append(nn.BatchNorm1d(hidden_size))\n                self.meta_model.append(nn.Dropout(0.10))\n                self.meta_model.append(activation)\n            else:\n                self.meta_model.append(nn.Linear(hidden_size, hidden_size))\n                #self.meta_model.append(nn.BatchNorm1d(hidden_size))\n                self.meta_model.append(nn.Dropout(0.10))\n                self.meta_model.append(activation)\n\n        self.linear = nn.Linear(hidden_size, num_output)\n\n        self.attention = attention\n        if self.attention == True:\n            self.att = nn.Sequential(\n                nn.Linear(input_size, hidden_size),\n                activation,\n                nn.Linear(hidden_size, input_size),\n                nn.Sigmoid()\n            )\n\n    def forward(self, x):\n\n        if self.attention==True:\n            w = self.att(x)\n            x = w*x\n        else:\n            x = x\n\n        global out\n        for i in range(len(self.meta_model)):\n            if i== 0:\n                out = self.meta_model[0](x)\n            else:\n                out = self.meta_model[i](out)\n\n        out = self.linear(out)\n\n        return out\n\nclass AMEX_Model(nn.Module):\n\n    def __init__(self, input_size, ffnn_input, hidden_size, \n                 num_layers, seq_length, activation, device=device):\n        \n        super(AMEX_Model, self).__init__()\n\n        self.device = device\n\n        self.encoders = torch.nn.ModuleList()\n        for i in range(seq_length):\n            self.encoders.append(MLP_MODEL(input_size=input_size,\n                                           hidden_size=hidden_size,\n                                           num_layers=num_layers,\n                                           num_output=64, \n                                           activation = activation))\n\n        self.ffnn = nn.Sequential(nn.Linear(ffnn_input, 256), \n                                        nn.Dropout(0.10),\n                                        activation,\n                                        nn.Linear(256, 256),\n                                        nn.Dropout(0.10),\n                                        activation,\n                                        nn.Linear(256, 256)\n                                        )\n        \n        \n        self.classifier = nn.Sequential(nn.Linear(64*seq_length+256+input_size, 256), \n                                        nn.Dropout(0.20),\n                                        activation,\n                                        nn.Linear(256, 128),\n                                        #nn.Dropout(0.10),\n                                        activation,\n                                        nn.Linear(128, 1), \n                                        nn.Sigmoid()\n                                        )\n        \n        \n\n    def forward(self, inp3D, inp2D):\n        \n        inp3D = inp3D.to(self.device)\n        inp2D = inp2D.to(self.device)\n        \n        encoded_input = []\n        for i in range(inp3D.shape[1]):\n            encoded_input.append(self.encoders[i](inp3D[:, i, :]))\n        \n        encoded_input = torch.cat(encoded_input, dim=1)\n        \n        x = self.ffnn(inp2D)\n        x = torch.cat([encoded_input, x, inp3D[:, -1, :]], dim=1)\n        \n        out = self.classifier(x)\n\n        return out","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:40:28.633871Z","iopub.execute_input":"2022-08-12T02:40:28.634245Z","iopub.status.idle":"2022-08-12T02:40:28.654056Z","shell.execute_reply.started":"2022-08-12T02:40:28.634208Z","shell.execute_reply":"2022-08-12T02:40:28.653059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### 1D CNN model #### \nclass CNN_AMEX(nn.Module):\n    \n    def __init__(self, input_size, ffnn_input, hidden_size, num_layers, \n                 seq_length, activation, device=device):\n        super().__init__()\n\n        self.num_layers = num_layers\n        self.input_size = input_size\n        self.hidden_size = hidden_size\n        self.seq_length = seq_length\n        self.device = device\n        \n        def _norm(layer, dim=None):\n            return nn.utils.weight_norm(layer, dim=dim) \n\n        self.conv1 = nn.Sequential(\n            nn.BatchNorm1d(seq_length),\n            #nn.Dropout(0.10),\n            nn.Conv1d(seq_length, 8, kernel_size=1, stride=1, bias=True),\n            activation,\n            #nn.AdaptiveAvgPool1d(output_size=128),\n            nn.BatchNorm1d(8),\n            #nn.Dropout(0.10),\n            nn.Conv1d(8, 4, kernel_size=1, stride=1, bias=True),\n            activation,\n            #nn.AdaptiveAvgPool1d(output_size=64)\n        )\n\n        self.flt = nn.Flatten()\n        \n        self.ffnn = nn.Sequential(nn.Linear(ffnn_input, 128), \n                                        nn.Dropout(0.10),\n                                        activation,\n                                        nn.Linear(128, 128),\n                                        nn.Dropout(0.10),\n                                        activation,\n                                        nn.Linear(128, 64)\n                                        )\n        \n\n        self.classifier = nn.Sequential(nn.Linear(5*input_size+64+ffnn_input, 256),\n                                        #nn.BatchNorm1d(512),\n                                        nn.Dropout(0.20),\n                                        activation,\n                                        nn.Linear(256, 128),\n                                        #nn.BatchNorm1d(256),\n                                        #nn.Dropout(0.20),\n                                        activation,\n                                        nn.Linear(128, 1), \n                                        nn.Sigmoid()\n                                        )\n \n\n    def forward(self, inp3D, inp2D):\n        \n        inp3D = inp3D.to(self.device)\n        inp2D = inp2D.to(self.device)\n\n        x1 = self.conv1(inp3D)\n        x1 = self.flt(x1)\n        \n        x2 = self.ffnn(inp2D)\n        x = torch.cat([x1, x2, inp3D[:, -1, :], inp2D], dim=1)\n        \n        out = self.classifier(x)\n\n        return out","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:40:28.655690Z","iopub.execute_input":"2022-08-12T02:40:28.656328Z","iopub.status.idle":"2022-08-12T02:40:28.669397Z","shell.execute_reply.started":"2022-08-12T02:40:28.656293Z","shell.execute_reply":"2022-08-12T02:40:28.668297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH_TO_DATA='../input/amex-sequence-data-preprocess/'","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:40:28.671051Z","iopub.execute_input":"2022-08-12T02:40:28.671572Z","iopub.status.idle":"2022-08-12T02:40:28.683247Z","shell.execute_reply.started":"2022-08-12T02:40:28.671538Z","shell.execute_reply":"2022-08-12T02:40:28.682236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class WeightedFocalLoss(nn.Module):\n    \"Weighted version of Focal Loss\"\n    def __init__(self, alpha=.25, gamma=2):\n        super(WeightedFocalLoss, self).__init__()\n        self.alpha = torch.tensor([alpha, 1-alpha]).cuda()\n        self.gamma = gamma\n\n    def forward(self, inputs, targets):\n        BCE_loss = F.binary_cross_entropy_with_logits(inputs, targets, reduction='none')\n        targets = targets.type(torch.long)\n        at = self.alpha.gather(0, targets.data.view(-1))\n        pt = torch.exp(-BCE_loss)\n        F_loss = at*(1-pt)**self.gamma * BCE_loss\n        return F_loss.mean()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:40:28.686065Z","iopub.execute_input":"2022-08-12T02:40:28.686370Z","iopub.status.idle":"2022-08-12T02:40:28.694733Z","shell.execute_reply.started":"2022-08-12T02:40:28.686347Z","shell.execute_reply":"2022-08-12T02:40:28.692889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def training_nn(Nfolds, MODEL_TYPE, lag, seq_len, num_epoch=50, patience=10,\n                num_layers=3, activation=nn.GELU(), MODEL_ROOT='models/lstm/',\n                hidden_dim=512):\n\n    uniques = {}\n\n    if not os.path.exists(MODEL_ROOT):\n        os.makedirs(MODEL_ROOT)\n\n    \n    scores = []\n    oof = {'customer_ID':[], 'target':[], 'oof':[]}\n       \n    for fold in range(Nfolds):\n       \n        model_path = MODEL_ROOT + f'/modeL_{fold}.pth'\n        folds = [0, 1, 2, 3, 4, 5, 6, 7, 8, 9]\n        \n        valid_idx = [fold, fold+Nfolds]\n        train_idx = [x for x in folds if x not in valid_idx]\n        \n        \n        # READ TRAIN DATA FROM DISK\n        X_train_3D = []; X_train_2D = []; y_train = []\n        for k in train_idx:\n            X_train_3D.append(np.load(f'{PATH_TO_DATA}train_num_{k}.npy'))\n            X_train_2D.append(np.concatenate((np.load(f'{PATH_TO_DATA}train_high_{k}.npy'), \n                                              np.load(f'{PATH_TO_DATA}train_skew_{k}.npy'), \n                                              #np.load(f'{PATH_TO_DATA}train_meduim_{k}.npy')\n                                             ), axis=1))\n            y_train.append( pd.read_pickle(f'{PATH_TO_DATA}targets_{k}.pkl') )\n        X_train_3D = np.concatenate(X_train_3D,axis=0)\n        X_train_2D = np.concatenate(X_train_2D,axis=0)                                  \n        y_train = pd.concat(y_train).target.values\n        \n    \n        #print('### Training data shapes', X_train_3D.shape, X_train_2D.shape, y_train.shape)\n\n        # READ VALID DATA FROM DISK\n        X_valid_3D = []; X_valid_2D = []; y_valid = []\n        for k in valid_idx:\n            X_valid_3D.append(np.load(f'{PATH_TO_DATA}train_num_{k}.npy'))\n            X_valid_2D.append(np.concatenate((np.load(f'{PATH_TO_DATA}train_high_{k}.npy'), \n                                              np.load(f'{PATH_TO_DATA}train_skew_{k}.npy'), \n                                              #np.load(f'{PATH_TO_DATA}train_meduim_{k}.npy')\n                                             ), axis=1))\n            y_valid.append(pd.read_pickle(f'{PATH_TO_DATA}targets_{k}.pkl') )\n        X_valid_3D = np.concatenate(X_valid_3D,axis=0)\n        X_valid_2D = np.concatenate(X_valid_2D,axis=0)                                  \n        \n        oof['customer_ID'] = oof['customer_ID'] + pd.concat(y_valid).customer_ID.to_list()\n        oof['target'] = oof['target'] + pd.concat(y_valid).target.to_list()\n        \n        y_valid = pd.concat(y_valid).target.values\n        #print('### Validation data shapes', X_valid_3D.shape, X_valid_2D.shape, y_valid.shape)\n\n        train_loader = AMEXLoader(X_train_3D, X_train_2D, y_train, lag, batch_size=1024, shuffle=True)\n        val_loader = AMEXLoader(X_valid_3D, X_valid_2D, y_valid, lag, batch_size=2048, shuffle=False)\n\n        del X_train_3D, X_train_2D, y_train\n        gc.collect()\n\n        if MODEL_TYPE=='CNN':\n            model = CNN_AMEX(input_size = X_valid_3D.shape[2], ffnn_input = X_valid_2D.shape[1], \n                             hidden_size=hidden_dim, num_layers=num_layers, seq_length=seq_len, \n                             activation=activation).to(device)\n            \n        elif MODEL_TYPE=='LSTM':\n            model = LSTM_AMEX(input_size = X_valid_3D.shape[2], ffnn_input = X_valid_2D.shape[1], \n                              hidden_size=hidden_dim, num_layers=num_layers, seq_length=seq_len, \n                              activation=activation).to(device)\n        else:\n            model = AMEX_Model(input_size = X_valid_3D.shape[2], ffnn_input = X_valid_2D.shape[1],\n                               hidden_size=256, num_layers=num_layers, seq_length=seq_len, \n                               activation=activation).to(device)\n\n        criterion = nn.BCELoss()\n\n        torch.manual_seed(42)\n        optimizer = torch.optim.AdamW(model.parameters(), lr=0.001, weight_decay=1e-5)\n        scheduler = optim.lr_scheduler.OneCycleLR(optimizer=optimizer, pct_start=0.1, div_factor=1e3,\n                                                  max_lr=2e-3, epochs=num_epoch, steps_per_epoch=len(train_loader))\n        #scheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer=optimizer, mode='min', patience=2, verbose=True, factor=0.1)\n\n\n        best_score=np.inf\n        best_y_pred = None\n        best_model=None\n        counter=0\n        for ep in range(num_epoch):\n            \n            train_loss, val_loss = 0, 0\n\n            model.train()\n            for X_3D, X_2D, y in tqdm(train_loader):\n              \n                optimizer.zero_grad()\n\n                out = model(X_3D, X_2D)\n                loss = criterion(out[:, 0], y.to(device))#+criterion(sim, y1.to(device))\n                \n                loss.backward()\n               \n                #torch.nn.utils.clip_grad_norm_(model.parameters(), 2)\n                optimizer.step()\n                scheduler.step()\n\n                with torch.no_grad():\n                    train_loss += loss.item() / len(train_loader)\n\n            # Validation phase\n            phase='Val'\n            with torch.no_grad():\n                model.eval()\n                \n#                 y_disc_true = []\n#                 y_disc_pred = []\n\n                y_true = []\n                y_pred = []\n                rloss = 0\n\n                for X_3D, X_2D, y in tqdm(val_loader):\n                    \n                    out = model(X_3D, X_2D)\n\n                    loss = criterion(out[:, 0], y.to(device))#+criterion(sim, y1.to(device))\n                    rloss += loss.item() / len(val_loader)\n                    \n#                     y_disc_pred += list(sim.sigmoid().detach().cpu().numpy().flatten())\n#                     y_disc_true += list(y1.cpu().numpy())\n                    \n                    y_pred += list(out.detach().cpu().numpy().flatten())\n                    y_true += list(y.cpu().numpy())\n                \n                 \n#                 y_disc_pred = np.round(y_disc_pred)\n#                 score_sim = accuracy_score(y_disc_true, y_disc_pred)\n                \n                score = amex_metric(y_true, y_pred)\n                if best_score>rloss:\n                    best_score=rloss\n                    best_y_pred = y_pred\n                    best_model=model\n                    torch.save(best_model, model_path)\n                    counter = 0\n                else:\n                    counter = counter+1\n\n\n                print(f\"[{phase}] Epoch: {ep} | Tain loss: {train_loss:.4f} | Val Loss: {rloss:.4f} | AMEX: {score:.4f} | Best AMEX: {best_score:.4f}\")\n\n                # plt.plot(y_true)\n                # plt.plot(y_pred)\n                # plt.show()\n                #scheduler.step(rloss)\n\n            if counter>=patience:\n                print(\"Early stopping\")\n                break\n        \n        \n        print(f'The best score - {fold}:', np.round(best_score, 6))\n        scores.append(best_score)\n        oof['oof'] = oof['oof'] + best_y_pred\n        del train_loader, val_loader, X_valid_3D, X_valid_2D, y_valid, model, best_model\n        gc.collect()\n\n\n\n    return scores, oof","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:40:28.698466Z","iopub.execute_input":"2022-08-12T02:40:28.699479Z","iopub.status.idle":"2022-08-12T02:40:28.725406Z","shell.execute_reply.started":"2022-08-12T02:40:28.699348Z","shell.execute_reply":"2022-08-12T02:40:28.724390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print(pd.read_pickle('../input/amex-sequence-data-preprocess/targets_0.pkl'))","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:40:28.726752Z","iopub.execute_input":"2022-08-12T02:40:28.727152Z","iopub.status.idle":"2022-08-12T02:40:28.739762Z","shell.execute_reply.started":"2022-08-12T02:40:28.727117Z","shell.execute_reply":"2022-08-12T02:40:28.738895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import matplotlib.pylab as plt\n# import seaborn as sns\n\n# for j in range(X_train.shape[2]):\n#     fig, (ax1, ax2, ax3, ax4, ax5, ax6, ax7, ax8, ax9, ax10, ax11, ax12, ax13) = plt.subplots(1, 13, figsize=(22, 3))\n#     for i, ax in zip(range(13), [ax1, ax2, ax3, ax4, ax5, ax6, ax7, ax8, ax9, ax10, ax11, ax12, ax13]):\n#         ax.hist(X_train[:, i, j])\n    \n#     plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:40:28.741277Z","iopub.execute_input":"2022-08-12T02:40:28.741686Z","iopub.status.idle":"2022-08-12T02:40:28.749681Z","shell.execute_reply.started":"2022-08-12T02:40:28.741597Z","shell.execute_reply":"2022-08-12T02:40:28.748567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lag=0\nfor MODEL_TYPE in ['MLP', 'CNN', 'LSTM']:\n    scores, oof = training_nn(Nfolds=5, MODEL_TYPE=MODEL_TYPE, lag=lag, \n                              seq_len=13-lag, num_epoch=20, patience=20,\n                              num_layers=2, activation = nn.CELU(), \n                              MODEL_ROOT=f'models/{MODEL_TYPE}/',\n                              hidden_dim=256)\n\n    print('Average score:', np.mean(scores))\n    print('OOF score:', amex_metric(oof['target'], oof['oof']))\n\n    oof = pd.DataFrame.from_dict(oof)\n    oof.to_csv(f'oof_{MODEL_TYPE}.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:40:28.751053Z","iopub.execute_input":"2022-08-12T02:40:28.751545Z","iopub.status.idle":"2022-08-12T02:54:38.547333Z","shell.execute_reply.started":"2022-08-12T02:40:28.751512Z","shell.execute_reply":"2022-08-12T02:54:38.546364Z"},"trusted":true},"execution_count":null,"outputs":[]}]}