{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport glob, gc","metadata":{"tags":[],"execution":{"iopub.status.busy":"2022-07-28T23:39:03.099209Z","iopub.execute_input":"2022-07-28T23:39:03.099998Z","iopub.status.idle":"2022-07-28T23:39:03.126914Z","shell.execute_reply.started":"2022-07-28T23:39:03.099893Z","shell.execute_reply":"2022-07-28T23:39:03.125939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reading Data","metadata":{}},{"cell_type":"code","source":"train = pd.read_parquet(\"../input/amex-fe/train_fe.parquet\")\n\n# train_label = pd.read_csv('../input/amex-default-prediction/train_labels.csv')\n# train_label = train_label.set_index('customer_ID')","metadata":{"tags":[],"execution":{"iopub.status.busy":"2022-07-28T23:39:03.147791Z","iopub.execute_input":"2022-07-28T23:39:03.148607Z","iopub.status.idle":"2022-07-28T23:39:16.784027Z","shell.execute_reply.started":"2022-07-28T23:39:03.148570Z","shell.execute_reply":"2022-07-28T23:39:16.783054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T23:39:16.786508Z","iopub.execute_input":"2022-07-28T23:39:16.787310Z","iopub.status.idle":"2022-07-28T23:39:16.900029Z","shell.execute_reply.started":"2022-07-28T23:39:16.787260Z","shell.execute_reply":"2022-07-28T23:39:16.898692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T23:39:16.901957Z","iopub.execute_input":"2022-07-28T23:39:16.902784Z","iopub.status.idle":"2022-07-28T23:39:16.937644Z","shell.execute_reply.started":"2022-07-28T23:39:16.902732Z","shell.execute_reply":"2022-07-28T23:39:16.936603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_dim = train.shape[1] - 2","metadata":{"execution":{"iopub.status.busy":"2022-07-28T23:39:16.940304Z","iopub.execute_input":"2022-07-28T23:39:16.940624Z","iopub.status.idle":"2022-07-28T23:39:16.946149Z","shell.execute_reply.started":"2022-07-28T23:39:16.940583Z","shell.execute_reply":"2022-07-28T23:39:16.944768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport copy\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\nfrom torch.optim.lr_scheduler import MultiStepLR\n\n\ndef amex_metric(y_true: np.array, y_pred: np.array) -> float:\n\n    # count of positives and negatives\n    n_pos = y_true.sum()\n    n_neg = y_true.shape[0] - n_pos\n\n    # sorting by descring prediction values\n    indices = np.argsort(y_pred)[::-1]\n    preds, target = y_pred[indices], y_true[indices]\n\n    # filter the top 4% by cumulative row weights\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n    four_pct_filter = cum_norm_weight <= 0.04\n\n    # default rate captured at 4%\n    d = target[four_pct_filter].sum() / n_pos\n\n    # weighted gini coefficient\n    lorentz = (target / n_pos).cumsum()\n    gini = ((lorentz - cum_norm_weight) * weight).sum()\n\n    # max weighted gini coefficient\n    gini_max = 10 * n_neg * (1 - 19 / (n_pos + 20 * n_neg))\n\n    # normalized weighted gini coefficient\n    g = gini / gini_max\n\n    return 0.5 * (g + d)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2022-07-28T23:39:16.948147Z","iopub.execute_input":"2022-07-28T23:39:16.949480Z","iopub.status.idle":"2022-07-28T23:39:18.692560Z","shell.execute_reply.started":"2022-07-28T23:39:16.949430Z","shell.execute_reply":"2022-07-28T23:39:18.691270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DenseModel(nn.Module):\n    def __init__(self, in_feats, repeat=1):\n        super(DenseModel, self).__init__()\n        self.l1 = nn.Linear(in_feats, 400, bias=True)\n        self.l2 = nn.Linear(1318, 100, bias=True)\n        self.l3 = nn.Linear(1418, 1, bias=True)\n        self.relu = nn.ReLU()\n        self.dropout = nn.Dropout(0.7)\n        self.bn1 = nn.BatchNorm1d(in_feats)\n        self.bn2 = nn.BatchNorm1d(200)\n\n    def forward(self, x):\n        x = self.bn1(x)\n\n        x1 = self.l1(x)\n        x1 = self.dropout(x1)\n        x1 = self.relu(x1)\n\n        x_c1 = torch.cat([x, x1], 1)\n        x2 = self.l2(x_c1)\n        x2 = self.dropout(x2)\n        x2 = self.relu(x2)\n\n        x_c2 = torch.cat([x, x1, x2], 1)\n        x3 = self.l3(x_c2)\n        return x3","metadata":{"tags":[],"execution":{"iopub.status.busy":"2022-07-28T23:39:18.694137Z","iopub.execute_input":"2022-07-28T23:39:18.694713Z","iopub.status.idle":"2022-07-28T23:39:18.706739Z","shell.execute_reply.started":"2022-07-28T23:39:18.694680Z","shell.execute_reply":"2022-07-28T23:39:18.705284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = torch.from_numpy(train.iloc[:, 1:-1].fillna(0).values.astype(np.float16))\nY = torch.from_numpy(train.iloc[:, -1].values)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2022-07-28T23:39:18.708408Z","iopub.execute_input":"2022-07-28T23:39:18.708836Z","iopub.status.idle":"2022-07-28T23:39:25.083159Z","shell.execute_reply.started":"2022-07-28T23:39:18.708804Z","shell.execute_reply":"2022-07-28T23:39:25.081706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pdb\ndef train_and_val(model, train_loader, val_loader, loss_fn, epoch, path):\n    best_metric = 0\n    for _ in range(epoch):\n        model.train()\n        train_pred = []\n        train_label = []\n        for batch_x, batch_y in train_loader:\n            \n            optimizer.zero_grad()\n            pred = model(batch_x.float())\n            loss = loss_fn(pred[:, 0], batch_y.float())\n            loss.backward()\n            optimizer.step()\n\n            pred = torch.sigmoid(pred)\n            pred = pred.data.cpu().numpy().reshape(-1)\n            train_pred.append(pred)\n            train_label.append(batch_y.data.cpu().numpy())\n            \n        val_pred = []\n        val_label = []\n        model.eval()\n        with torch.no_grad():\n            for batch_x, batch_y in val_loader:\n                pred = torch.sigmoid(model(batch_x.float()))\n                pred = pred.data.cpu().numpy().reshape(-1)\n                val_pred.append(pred)\n                val_label.append(batch_y.data.cpu().numpy())\n        \n        # pdb.set_trace()\n        train_metric = amex_metric(np.hstack(train_label), np.hstack(train_pred))\n        val_metric = amex_metric(np.hstack(val_label), np.hstack(val_pred))\n\n        if best_metric < val_metric:\n            best_metric = val_metric\n            torch.save(model.state_dict(), path)\n            print(best_metric, 'Saving to', path)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2022-07-28T23:39:25.084800Z","iopub.execute_input":"2022-07-28T23:39:25.085128Z","iopub.status.idle":"2022-07-28T23:39:25.099719Z","shell.execute_reply.started":"2022-07-28T23:39:25.085096Z","shell.execute_reply":"2022-07-28T23:39:25.097895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\n\nskf = StratifiedKFold(n_splits=6)\nloss_fn = torch.nn.BCEWithLogitsLoss()\n\nepoch = 10\nbagging = 5\n\nfor fold_idx, (tr_idx, val_idx) in enumerate(skf.split(X, Y)):\n    print(tr_idx, val_idx)\n    fold_tr_dataset = torch.utils.data.TensorDataset(X[tr_idx], Y[tr_idx])\n    fold_val_dataset = torch.utils.data.TensorDataset(X[val_idx], Y[val_idx])\n    val_loader = torch.utils.data.DataLoader(fold_val_dataset, batch_size=1024, shuffle=False)\n\n    for bag_idx in range(bagging):\n        bagging_idx = np.random.choice(range(len(fold_tr_dataset)),  int(len(fold_tr_dataset)*0.9))\n        fold_tr_bagging_dataset = torch.utils.data.Subset(fold_tr_dataset, bagging_idx)\n\n        train_loader = torch.utils.data.DataLoader(fold_tr_bagging_dataset, batch_size=1024*4, shuffle=True)\n\n        model = DenseModel(feature_dim)\n        optimizer = torch.optim.Adam(model.parameters(), lr=0.001)\n        train_and_val(model, train_loader, val_loader, loss_fn, epoch, f'dense_{fold_idx}_{bag_idx}.pt')\n    \n        del fold_tr_bagging_dataset, train_loader, model, optimizer\n        # break","metadata":{"scrolled":true,"tags":[],"execution":{"iopub.status.busy":"2022-07-28T23:39:25.101875Z","iopub.execute_input":"2022-07-28T23:39:25.102738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X, Y;\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T17:42:55.070929Z","iopub.status.idle":"2022-07-26T17:42:55.071703Z","shell.execute_reply.started":"2022-07-26T17:42:55.071471Z","shell.execute_reply":"2022-07-26T17:42:55.071493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}