{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport gc\nimport pickle\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport torch\n\nimport torch.nn as nn\nimport matplotlib.pyplot as plt\n\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import f1_score","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-18T16:45:52.633135Z","iopub.execute_input":"2022-08-18T16:45:52.633563Z","iopub.status.idle":"2022-08-18T16:45:52.640586Z","shell.execute_reply.started":"2022-08-18T16:45:52.633528Z","shell.execute_reply":"2022-08-18T16:45:52.639258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device=torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(device)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:45:52.643541Z","iopub.execute_input":"2022-08-18T16:45:52.644084Z","iopub.status.idle":"2022-08-18T16:45:52.654168Z","shell.execute_reply.started":"2022-08-18T16:45:52.644033Z","shell.execute_reply":"2022-08-18T16:45:52.653070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Config","metadata":{}},{"cell_type":"code","source":"class CFG:\n    BATCH_SIZE=2048\n    N_EPOCHS=8","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:45:52.656268Z","iopub.execute_input":"2022-08-18T16:45:52.657309Z","iopub.status.idle":"2022-08-18T16:45:52.667161Z","shell.execute_reply.started":"2022-08-18T16:45:52.657248Z","shell.execute_reply":"2022-08-18T16:45:52.666049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# load dataset","metadata":{}},{"cell_type":"code","source":"%%time\ntrain_customer_ids = np.load(\"../input/amex-sequence-dataset-v3/train_sequence/all_train_customerIds.npy\")\n\nxseq = np.load(\"../input/amex-sequence-dataset-v3/train_sequence/all_train_num_feats.npy\")\nxseq_cat = np.load(\"../input/amex-sequence-dataset-v3/train_sequence/all_train_cat_feats.npy\")\nxseq = np.clip(xseq, -3, 3)\n\nytrain = np.load(\"../input/amex-sequence-dataset-v3/train_sequence/all_targets.npy\")\n\n\n#Tabular dataset loading\nxtab = np.load(\"../input/amex-tabular-train-nn-dataset-v3/train_data/x_train.npy\")\nxtab_miss = np.load(\"../input/amex-tabular-train-nn-dataset-v3/train_data/x_miss_train.npy\")\nxtab_count = np.load(\"../input/amex-tabular-train-nn-dataset-v3/train_data/x_count.npy\")\nxtab = np.clip(xtab, -6.0, 6.0)\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:45:52.668435Z","iopub.execute_input":"2022-08-18T16:45:52.669589Z","iopub.status.idle":"2022-08-18T16:46:43.660555Z","shell.execute_reply.started":"2022-08-18T16:45:52.669547Z","shell.execute_reply":"2022-08-18T16:46:43.659214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(xtab.shape, xseq.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:46:43.663163Z","iopub.execute_input":"2022-08-18T16:46:43.664214Z","iopub.status.idle":"2022-08-18T16:46:43.671058Z","shell.execute_reply.started":"2022-08-18T16:46:43.664170Z","shell.execute_reply":"2022-08-18T16:46:43.669833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataset","metadata":{}},{"cell_type":"code","source":"class AmexDataset(torch.utils.data.Dataset):\n    def __init__(self, xtab, xtab_count, xtab_miss, xseq, xseq_cat, y, idxs, phase='train'):\n        self.idxs=idxs\n        self.y = y\n        self.xtab=xtab\n        self.xtab_count=xtab_count\n        self.xtab_miss=xtab_miss\n        \n        self.xseq=xseq\n        self.xseq_cat=xseq_cat\n        \n        self.phase=phase\n    \n    def get_tab_features(self, idx):\n        x = torch.tensor(self.xtab[idx], dtype=torch.float32)\n        xmissing = 1+(x==-1).type(torch.long)\n        x_count = torch.tensor(self.xtab_count[idx]/13, dtype=torch.float32)\n        xmiss_count = torch.tensor(self.xtab_miss[idx]/13, dtype=torch.float32)\n        \n        return {\n            'x': x,\n            'xmissing': xmissing,\n            'xcount': x_count,\n            'xmiss_count': xmiss_count\n        }\n    \n    def get_seq_features(self, idx):\n        x = torch.tensor(self.xseq[idx], dtype=torch.float32)\n        xmissing = 1+(x==-1).type(torch.long)\n        xcat = torch.tensor(self.xseq_cat[idx], dtype=torch.long)\n        return {\n            'x': x,\n            'xmissing': xmissing,\n            'xcat': xcat\n        }\n    def __getitem__(self, idx):\n        idx = self.idxs[idx]\n        tabfeats = self.get_tab_features(idx)\n        seqfeats = self.get_seq_features(idx)\n        y = torch.tensor(self.y[idx], dtype=torch.float32)\n        \n        return tabfeats, seqfeats, y\n    \n    def __len__(self):\n        return len(self.idxs)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:46:43.672609Z","iopub.execute_input":"2022-08-18T16:46:43.672991Z","iopub.status.idle":"2022-08-18T16:46:43.688433Z","shell.execute_reply.started":"2022-08-18T16:46:43.672957Z","shell.execute_reply":"2022-08-18T16:46:43.686848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tabular model","metadata":{}},{"cell_type":"code","source":"class TransformBlock(nn.Module):\n    def __init__(self, insize, outsize, dropout=0.1):\n        super().__init__()\n        self.bn = nn.BatchNorm1d(insize)\n        self.linear = nn.Linear(insize, outsize)\n        self.activation = nn.Softplus()\n        self.dropout = nn.Dropout(dropout)\n    \n    def forward(self, x):\n        x = self.bn(x)\n        x = self.linear(x)\n        x = self.activation(x)\n        x = self.dropout(x)\n        return x\n    \nclass TabularModel(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.pre_bn = nn.BatchNorm1d(1401)\n        self.missing_embedd = nn.Embedding(3, 5)\n        self.preprocess_layer = nn.Linear(6, 1)\n        self.dropout=nn.Dropout(0.01)\n        \n        self.layer1 = TransformBlock(1401, 1024, dropout = 0.5)\n        self.layer2 = TransformBlock(1024, 512, dropout = 0.5)\n        self.layer3 = TransformBlock(512, 256, dropout = 0.5)\n        \n        self.head = nn.Sequential(\n            nn.Linear(256, 1)\n        )\n        \n    def forward(self, tabfeats):\n        x=tabfeats['x']\n        xmissing=tabfeats['xmissing']\n        x_count=tabfeats['xcount']\n        xmiss_count=tabfeats['xmiss_count']\n        \n        xmissing = self.missing_embedd(xmissing)\n        x = x.unsqueeze(dim=-1)\n        x = torch.cat([x, xmissing], dim=-1)\n        \n        xmeta = torch.cat([x_count.unsqueeze(dim=-1), xmiss_count], dim=-1)\n        x = self.preprocess_layer(x).squeeze(dim=-1)\n        x = torch.cat([x, xmeta], dim=-1)\n        \n        x = self.dropout(x)\n        \n        x1 = self.layer1(x)\n        x2 = self.layer2(x1)\n        x3 = self.layer3(x2)\n        \n        return (x2, x3)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:46:43.690025Z","iopub.execute_input":"2022-08-18T16:46:43.690427Z","iopub.status.idle":"2022-08-18T16:46:43.708835Z","shell.execute_reply.started":"2022-08-18T16:46:43.690388Z","shell.execute_reply":"2022-08-18T16:46:43.707772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sequence Model","metadata":{}},{"cell_type":"code","source":"class ProjectionBlock(nn.Module):\n    def __init__(self, in_size, out_size):\n        super().__init__()\n        self.dropout = nn.Dropout(0.5)\n        self.bn = nn.BatchNorm1d(in_size)\n        self.linear = nn.Linear(in_size, out_size)\n        self.activation = nn.Softplus()\n        \n        \n    def forward(self, x):\n        x = self.dropout(x)\n        x = self.bn(x)\n        x = self.linear(x)\n        x = self.activation(x)\n        return x\n\nclass ProjectionMLP(nn.Module):\n    def __init__(self, sz):\n        super().__init__()\n        self.proj1 = ProjectionBlock(sz, sz)\n        self.proj2 = ProjectionBlock(sz, sz//2)\n        self.proj3 = ProjectionBlock(sz//2, sz//4)\n        \n        self.dropout = nn.Dropout(0.5)\n        self.bn = nn.BatchNorm1d(sz//4)\n        self.out = nn.Linear(sz//4, 1)\n        \n        \n        self.proj1.linear.weight.data.uniform_(-0.03, 0.03)\n        self.proj1.linear.bias.data.uniform_(-0.03, 0.03)\n        \n        self.proj2.linear.weight.data.uniform_(-0.03, 0.03)\n        self.proj2.linear.bias.data.uniform_(-0.03, 0.03)\n        \n        self.proj3.linear.weight.data.uniform_(-0.03, 0.03)\n        self.proj3.linear.bias.data.uniform_(-0.03, 0.03)\n        \n    def forward(self, x):\n        x = self.proj1(x)\n        x = self.proj2(x)\n        x = self.proj3(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2022-08-18T17:28:30.432701Z","iopub.execute_input":"2022-08-18T17:28:30.433183Z","iopub.status.idle":"2022-08-18T17:28:30.446708Z","shell.execute_reply.started":"2022-08-18T17:28:30.433146Z","shell.execute_reply":"2022-08-18T17:28:30.445290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class AmexGruModel(nn.Module):\n    def __init__(self):\n        super().__init__()\n        DIM = 256\n        \n        self.embeddings = nn.ModuleList([nn.Embedding(10, 4, padding_idx=0) for _ in range(13)])    \n        self.time_embedding = nn.Embedding(13, 5)\n        \n        self.missing_embedd = nn.Embedding(3, 5)\n        self.preprocess_layer = nn.Linear(6, 1)\n        \n        self.pre_ln = nn.LayerNorm(175)\n        self.gru = nn.GRU(175+52, DIM, bidirectional = True, batch_first = True)\n        self.activation = nn.ReLU6()\n        \n        self.mlp = ProjectionMLP(2*DIM)\n        self.mlp0 = ProjectionMLP(DIM)\n        self.mlp1 = ProjectionMLP(DIM)\n        \n        self.embedd_out =  nn.Sequential(\n            nn.Dropout(0.5),\n            nn.Linear(2*DIM, 1)\n        )\n        \n        self.embedd_out0 =  nn.Sequential(\n            nn.Dropout(0.5),\n            nn.Linear(DIM, 1)\n        )\n        \n        self.embedd_out1 =  nn.Sequential(\n            nn.Dropout(0.5),\n            nn.Linear(DIM, 1)\n        )\n        \n        \n        for n,p in self.gru.named_parameters():\n            torch.nn.init.uniform_(p.data, -0.04, 0.04)\n    \n    def precomputewithmissingvalues(self, x, xmissing):\n        x_list = []\n        for i in range(13):\n            xi = x[:, i].unsqueeze(dim=-1)\n            xmissing_i = self.missing_embedd(xmissing[:, i])\n            xi = torch.cat([xi, xmissing_i], dim=-1)\n            xi = self.preprocess_layer(xi).squeeze(dim=-1).unsqueeze(dim=1)\n            x_list.append(xi)\n        x = torch.cat(x_list, dim=1)\n        return x\n    \n    def concat_categorical_embeddings(self, x, xcat):\n        xcat_embedds = []\n        for i in range(13):\n            xcat_embedds.append( self.embeddings[i](xcat[:, :, i]) )\n        xcat_embedds = torch.cat(xcat_embedds, dim=-1)\n        x = torch.cat([x, xcat_embedds], dim=-1)\n        return x\n    \n    def forward(self, seqfeats):\n        x = seqfeats['x']\n        xmissing= seqfeats['xmissing']\n        xcat = seqfeats['xcat']\n        \n        x = self.precomputewithmissingvalues(x, xmissing)\n        x = self.pre_ln(x)\n        x = self.concat_categorical_embeddings(x, xcat)\n        \n        (x, h) =  self.gru(x)\n        h = self.activation(h)\n        \n        #Embeddings\n        h0 = h[0]\n        h1 = h[1]\n        \n        z1 = torch.cat([h0, h1], dim=-1)\n        z2 = self.mlp(z1)\n        return (z1, z2)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T17:28:30.449870Z","iopub.execute_input":"2022-08-18T17:28:30.450341Z","iopub.status.idle":"2022-08-18T17:28:30.471378Z","shell.execute_reply.started":"2022-08-18T17:28:30.450213Z","shell.execute_reply":"2022-08-18T17:28:30.470050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class AmexModel(nn.Module):\n    def __init__(self, tabpath, grupath):\n        super().__init__()\n        self.tabmodel = torch.load(tabpath, map_location=device)\n        self.grumodel = torch.load(grupath, map_location=device)\n        self.dropout = nn.Dropout(0.5)\n        \n        self.mlp = nn.Sequential(\n            nn.BatchNorm1d(2*256+256),\n            nn.Linear(2*256+256, 1024),\n            nn.Softplus(),\n            nn.Dropout(0.5),\n            \n            nn.BatchNorm1d(1024),\n            nn.Linear(1024, 512),\n            nn.Softplus(),\n            nn.Dropout(0.5),\n            \n            nn.BatchNorm1d(512),\n            nn.Linear(512, 256),\n            nn.Softplus(),\n            nn.Dropout(0.5)\n        )\n        \n        self.out = nn.Linear(256, 1)\n        for n,p in self.mlp.named_parameters():\n            torch.nn.init.uniform_(p.data, -0.02, 0.02)\n            \n    def forward(self, tabfeats, seqfeats):\n        self.tabmodel.eval()\n        self.grumodel.eval()\n        \n        with torch.no_grad():\n            (ztab1, ztab2) = self.tabmodel(tabfeats)\n            (zseq1, zseq2) = self.grumodel(seqfeats)\n\n        #Since the tabfeature is already dropout happens\n        #zseq = torch.cat([zseq1, zseq2], dim=-1)\n        \n        z = torch.cat([ztab2, zseq1], dim=-1)\n        z = self.dropout(z)\n        z = self.mlp(z)\n        \n        y = self.out(z).view(-1)\n        return y","metadata":{"execution":{"iopub.status.busy":"2022-08-18T17:28:30.485180Z","iopub.execute_input":"2022-08-18T17:28:30.486303Z","iopub.status.idle":"2022-08-18T17:28:30.500528Z","shell.execute_reply.started":"2022-08-18T17:28:30.486250Z","shell.execute_reply":"2022-08-18T17:28:30.499221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# metrics","metadata":{}},{"cell_type":"code","source":"def top_4percent(pred_df):\n    df = pred_df.copy()\n    df = df.sort_values('pred', ascending=False)\n    df['weight'] = df['target'].apply(lambda v: 20 if v==0 else 1)\n    four_percent_cutoff = 0.04 * sum(df['weight'])\n    df['weight_cumsum'] = df['weight'].cumsum()\n    df_cutoff = df[df.weight_cumsum <= four_percent_cutoff]\n    \n    return df_cutoff['target'].sum()/df['target'].sum()\n\ndef weighted_gini(pred_df):\n    df = pred_df.copy()\n    df = df.sort_values('pred', ascending=False)\n    df['weight'] = df['target'].apply(lambda v: 20 if v==0 else 1)\n    df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n    total_pos = (df['target'] * df['weight']).sum()\n    df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n    df['lorentz'] = df['cum_pos_found'] / total_pos\n    df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n    return df['gini'].sum()\n\n\ndef normalized_gini(df):\n    df_true=df[['target']].copy()\n    df_true['pred'] = df_true['target'].copy()\n    \n    G = weighted_gini(df)/weighted_gini(df_true)\n    return G","metadata":{"execution":{"iopub.status.busy":"2022-08-18T17:28:30.502249Z","iopub.execute_input":"2022-08-18T17:28:30.502676Z","iopub.status.idle":"2022-08-18T17:28:30.518634Z","shell.execute_reply.started":"2022-08-18T17:28:30.502641Z","shell.execute_reply":"2022-08-18T17:28:30.517288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# evaluate","metadata":{}},{"cell_type":"code","source":"def evaluate(foldnum, val_index, model, val_dataloader):\n    model.eval()\n    ytrue=[]\n    ypred=[]\n    \n    for (tabfeats, seqfeats, y) in val_dataloader:\n        for k,v in tabfeats.items():\n            tabfeats[k] = v.to(device)\n        for k,v in seqfeats.items():\n            seqfeats[k] = v.to(device)\n        y = y.to(device)\n        \n        with torch.no_grad():\n            yhat = model(tabfeats, seqfeats)\n            yhat = yhat.sigmoid()\n            ytrue += y.cpu().tolist()\n            ypred += yhat.cpu().tolist()\n    \n    df = pd.DataFrame.from_dict({\n        'customer_ids': train_customer_ids[val_index],\n        'target': ytrue,\n        'pred': ypred\n    })\n    df['predlabel'] = (df['pred'] > 0.5).astype(int)\n    \n    \n    print(\"====================================================\")\n    ypred0 = (df[df.target==0].pred).mean()\n    ypred1 = (df[df.target==1].pred).mean()\n    \n    print(\"avg non-defaulter prob:{:.4f}\".format(ypred0))\n    print(\"avg defaulter prob:{:.4f}\".format(ypred1))\n    \n    print(\"f1_score:{:.4f}\".format(f1_score(df.target, df.predlabel)))\n    print(\"proportion of non defaulter >0.3: {:.4f}\".format(len(df[(df.target==0) & (df.pred>=0.3)])/len(df) ))\n    print(\"proportion of defaulter < 0.7: {:.4f}\".format(len(df[(df.target==1) & (df.pred <= 0.7)])/len(df) ))\n    print(\"====================================================\")\n    print()\n    \n    \n    \n    df.to_csv(\"eval_{}.csv\".format(foldnum))\n    G = normalized_gini(df[['target', 'pred']])\n    D = top_4percent(df[['target', 'pred']])\n    M = (G+D)/2\n    \n    return (G, D, M)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T17:28:32.546661Z","iopub.execute_input":"2022-08-18T17:28:32.547187Z","iopub.status.idle":"2022-08-18T17:28:32.561735Z","shell.execute_reply.started":"2022-08-18T17:28:32.547150Z","shell.execute_reply":"2022-08-18T17:28:32.560777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# train ops","metadata":{}},{"cell_type":"code","source":"def get_rank_loss(yhat, y):\n    loss = torch.tensor(0.0, device=device)\n    ypos = yhat[y==1]\n    yneg = yhat[y==0]\n    \n    if len(ypos) == 0 or len(yneg) == 0:\n        return loss\n    \n    yneg = yneg.repeat((len(ypos), 1))\n    ypos = ypos.unsqueeze(dim=-1)\n    loss1 = -torch.log( torch.sigmoid( ypos.detach()-yneg) ).mean()\n    loss2 = -torch.log( torch.sigmoid( ypos-yneg.detach()) ).mean()\n    loss = (loss1+loss2)/2\n    return loss\n\ndef get_hinge_loss(yhat, y):\n    yhat = torch.clamp(yhat, -3, 3)\n    yerr = y*(1 - yhat) + (1-y) * (1+yhat)\n    yerr = torch.clamp(yerr, 0, 3)\n    loss = torch.mean(yerr)\n    return loss\n\ndef get_klregularization(y, yhat):\n    loss = torch.tensor(0.0, device=device)\n    ypos = yhat[y==1]\n    yneg = yhat[y==0]\n    cnt=0\n    if len(ypos) > 0:\n        cnt+=1\n        loss = loss - 2*torch.log(1e-9+ypos.sigmoid().mean())\n    if len(yneg) > 0:\n        cnt+=1\n        loss = loss - torch.log(1e-9+1-yneg.sigmoid().mean())\n    loss = loss/max(1, cnt)\n    return loss","metadata":{"execution":{"iopub.status.busy":"2022-08-18T17:28:34.533055Z","iopub.execute_input":"2022-08-18T17:28:34.533466Z","iopub.status.idle":"2022-08-18T17:28:34.546772Z","shell.execute_reply.started":"2022-08-18T17:28:34.533432Z","shell.execute_reply":"2022-08-18T17:28:34.545449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_ops(tabfeats, seqfeats, y, model, criterion, optimizer, scheduler):\n    y = y.to(device)\n    for k,v in tabfeats.items():\n        tabfeats[k] = v.to(device)\n\n    for k,v in seqfeats.items():\n        seqfeats[k] = v.to(device)\n    \n    model.train()\n    yhat = model(tabfeats, seqfeats)\n    \n    binary_loss = criterion(yhat, y)\n    rank_loss = get_rank_loss(yhat, y)\n    hinge_loss = get_hinge_loss(yhat, y)\n    \n    loss  = binary_loss + rank_loss + hinge_loss\n    \n    optimizer.zero_grad(set_to_none=True)\n    loss.backward()\n    torch.nn.utils.clip_grad_norm_(model.parameters(), 5)\n    optimizer.step()\n    scheduler.step()\n    return {\n        'loss':loss.item(),\n        'binary_loss': binary_loss.item(),\n        'rank_loss': rank_loss.item(),\n        'hinge_loss': hinge_loss.item()\n    }","metadata":{"execution":{"iopub.status.busy":"2022-08-18T17:28:36.793539Z","iopub.execute_input":"2022-08-18T17:28:36.794029Z","iopub.status.idle":"2022-08-18T17:28:36.805047Z","shell.execute_reply.started":"2022-08-18T17:28:36.793987Z","shell.execute_reply":"2022-08-18T17:28:36.803251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# train model","metadata":{}},{"cell_type":"code","source":"def train_model(s, foldnum, val_index, train_dataloader, val_dataloader):\n    grupath = \"../input/amex-gru-mean-feats-models/models/model_0_{}.pt\".format(foldnum)\n    tabpath = \"../input/amex-nn-ranking-1024-models/models/tabular_model_0_{}.pt\".format(foldnum)\n    \n    model = AmexModel(tabpath, grupath).to(device)\n    criterion = nn.BCEWithLogitsLoss()\n    mse_loss = nn.MSELoss()\n    \n    param_groups=[]\n    for n, p in model.named_parameters():\n        if n.startswith(\"tabmodel\") or n.startswith(\"grumodel\"):\n            param_groups.append({'params': p,'lr': 1e-6, 'weight_decay': 0.01})\n        else:\n            param_groups.append({'params': p,'lr': 1e-3, 'weight_decay': 1e-3})\n        \n    optimizer = torch.optim.AdamW(param_groups, lr=1e-3, weight_decay=1e-8)\n    scheduler = torch.optim.lr_scheduler.CosineAnnealingLR(optimizer, \n                                                           T_max = CFG.N_EPOCHS * len(train_dataloader), \n                                                           eta_min=1e-7)\n    \n    \n    best_eval=None\n    for e in range(CFG.N_EPOCHS):\n        model.train()\n        epoch_loss=[]\n        epoch_binary_loss=[]\n        epoch_rank_loss=[]\n        epoch_hinge_loss=[]\n        \n        \n        for it, (tabfeats, seqfeats, y) in enumerate(train_dataloader):\n            losses = train_ops(tabfeats, seqfeats, y, model, criterion, optimizer, scheduler)\n            epoch_loss.append(losses['loss'])\n            epoch_binary_loss.append(losses['binary_loss'])\n            epoch_rank_loss.append(losses['rank_loss'])\n            epoch_hinge_loss.append(losses['hinge_loss'])\n        \n        (G, D, M) = evaluate(foldnum, val_index, model, val_dataloader)\n        if best_eval is None or best_eval<M:\n            best_eval = M\n            torch.save(model, \"models/model{}_{}.pt\".format(s, foldnum))\n        \n        print(\"epoch:{} | loss:{:.4f}\".format(e, np.mean(epoch_loss)))\n        print(\"rank loss:{:.4f}\".format(np.mean(epoch_rank_loss)))\n        print(\"binary loss:{:.4f}\".format(np.mean(epoch_binary_loss)))\n        print(\"hinge loss:{:.4f}\".format(np.mean(epoch_hinge_loss)))\n        \n        print(\"current Eval:{:.4f} | best Eval:{:.4f}\".format(M, best_eval))\n        print(\"Gini:{:.4f} | Default Rate:{:4f}\".format(G, D))\n    \n    print()\n    print()\n    print()\n    print(\"Best Eval at the end of foldnumber:{} : {:.6f}\".format(foldnum, best_eval))","metadata":{"execution":{"iopub.status.busy":"2022-08-18T17:28:47.578538Z","iopub.execute_input":"2022-08-18T17:28:47.579021Z","iopub.status.idle":"2022-08-18T17:28:47.595728Z","shell.execute_reply.started":"2022-08-18T17:28:47.578980Z","shell.execute_reply":"2022-08-18T17:28:47.594640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not os.path.exists('models'):\n    os.mkdir(\"models\")","metadata":{"execution":{"iopub.status.busy":"2022-08-18T17:28:50.407213Z","iopub.execute_input":"2022-08-18T17:28:50.408391Z","iopub.status.idle":"2022-08-18T17:28:50.413369Z","shell.execute_reply.started":"2022-08-18T17:28:50.408347Z","shell.execute_reply":"2022-08-18T17:28:50.412203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skf = StratifiedKFold(n_splits=5, random_state=88471, shuffle=True)\n\nfor foldnum, (train_index, val_index) in enumerate(skf.split(ytrain, ytrain)):\n    train_dataset = AmexDataset(xtab, xtab_count, xtab_miss, xseq, xseq_cat, ytrain, train_index)\n    val_dataset = AmexDataset(xtab, xtab_count, xtab_miss, xseq, xseq_cat, ytrain, val_index)\n\n    train_dataloader = torch.utils.data.DataLoader(train_dataset, \n                                                   batch_size=CFG.BATCH_SIZE, \n                                                   shuffle=True,\n                                                   drop_last=True)\n\n    val_dataloader = torch.utils.data.DataLoader(val_dataset, batch_size=CFG.BATCH_SIZE, \n                                                   shuffle=False,\n                                                   drop_last=False)\n\n\n    print(\"Foldnumber:\", foldnum)\n    print(\"number of train iterations:\", len(train_dataloader))\n    print(\"number of val iterations:\", len(val_dataloader))\n\n    for s in range(1):\n        train_model(s, foldnum, val_index, train_dataloader, val_dataloader)\n        model = torch.load(\"models/model{}_{}.pt\".format(s, foldnum), map_location=device)\n        (G, D, M) = evaluate(foldnum, val_index, model, val_dataloader)\n        print(\"End of foldnumber:{} | Seed:{}\".format(foldnum, s))","metadata":{"execution":{"iopub.status.busy":"2022-08-18T17:28:52.096114Z","iopub.execute_input":"2022-08-18T17:28:52.096553Z","iopub.status.idle":"2022-08-18T17:28:58.486267Z","shell.execute_reply.started":"2022-08-18T17:28:52.096516Z","shell.execute_reply":"2022-08-18T17:28:58.484667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}