{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Time Series GRU PyTorch-ignite Training Notebook\nThis notebook is based on implementation of the [TensorFlow GRU Starter - [0.790]](https://www.kaggle.com/code/cdeotte/tensorflow-gru-starter-0-790/)\nand [AMEX PyTorch GRU: Training](https://www.kaggle.com/code/voix97/amex-pytorch-gru-training/notebook)\n\n* Data has been preprocessed , name`gru_data.ftr`\n* Less code\n* Base on `pytorch-ignite` , you can using the engine to add function in  `[EPOCH_COMPLETED(start)、ITERATION_COMPLETED(start)]`\n* Output `Amex score` while epoch is completed\n* Using 5 `StratifiedKFold`\n* add scheduler`(ReduceLROnPlateauScheduler)`\n\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.optim import Adam, AdamW\nfrom torch.optim.lr_scheduler import MultiStepLR, ReduceLROnPlateau\nfrom torch.utils.data import DataLoader, TensorDataset\nfrom torch.utils.data.dataset import Subset\nfrom ignite.engine import create_supervised_evaluator, create_supervised_trainer\nfrom ignite.engine import Events\nfrom ignite.metrics import Accuracy, RunningAverage, Loss\nfrom ignite.contrib.handlers import ProgressBar\nfrom ignite.handlers.param_scheduler import LRScheduler\nfrom sklearn.model_selection import KFold, StratifiedKFold\nfrom ignite.handlers.param_scheduler import ReduceLROnPlateauScheduler\n\nimport pandas as pd\nimport numpy as np\nimport time\nimport os ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cuda = False\ndevice = torch.device(\"cuda\" if cuda else \"cpu\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Gru_nn(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.gru = nn.GRU(input_size=188, hidden_size=256, batch_first=True,  bidirectional=True)\n        self.fc = nn.Sequential(nn.Linear(256, 64),\n                                nn.ReLU(),\n                                nn.Linear(64, 32),\n                                nn.ReLU(),\n                                nn.Linear(32, 2))\n    \n    def forward(self, x):\n        _, hidden = self.gru(x)\n        x = hidden[-1, :]\n        x = self.fc(x)\n        return x\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_feather(\"gru_data.ftr\")\ntargets = np.array(df.target).reshape(-1, 13)[:, -1]\ndf = np.array(df.iloc[:, 1:-1]).reshape(-1, 13, 188)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric(y_true, y_pred):\n\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n    print(\"G: {:.6f}, D: {:.6f}, ALL: {:6f}\".format(gini[1]/gini[0], top_four, 0.5*(gini[1]/gini[0] + top_four)))\n    return 0.5 * (gini[1]/gini[0] + top_four)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kfold = StratifiedKFold(n_splits = 5, shuffle = True, random_state = 42)\nfor fold, (trn_ind, val_ind) in enumerate(kfold.split(df, targets)):\n    print(' ')\n    print('-'*50)\n    print(f'Training fold {fold} ...')\n    x_train, x_val = df[trn_ind], df[val_ind]\n    y_train, y_val = targets[trn_ind], targets[val_ind]\n\n    x_train = torch.tensor(np.array(x_train, dtype=np.float32))\n    y_train = torch.LongTensor(np.array(y_train))\n    x_val = torch.tensor(np.array(x_val, dtype=np.float32))\n    y_val = torch.LongTensor(np.array(y_val))\n    \n    train_loader = DataLoader(TensorDataset(x_train, y_train), batch_size=64, shuffle=True)\n    val_loader = DataLoader(TensorDataset(x_val, y_val), batch_size=128, shuffle=False)\n    \n    model = Gru_nn().to(device)\n    optimizer = Adam(model.parameters(), lr=0.01)\n\n    trainer = create_supervised_trainer(model, optimizer, F.cross_entropy, device=device)\n    evaluator = create_supervised_evaluator(model, metrics={\"accuracy\":  Accuracy(), \"loss\": Loss(nn.CrossEntropyLoss())}, device=device)\n    \n    scheduler = ReduceLROnPlateauScheduler(optimizer, \"loss\",\n        save_history=True, mode=\"min\",\n        factor=0.5, patience=3, threshold_mode='rel',\n        threshold=0.1, trainer=trainer)\n\n    trainer.add_event_handler(Events.EPOCH_COMPLETED, scheduler)\n\n    pbar = ProgressBar(persist=False)\n    pbar.attach(trainer)\n\n    RunningAverage(output_transform=lambda x:x).attach(trainer, \"loss\")\n\n    all_pred = []\n    all_true = []\n    score_list = []\n\n    @trainer.on(Events.EPOCH_COMPLETED)\n    def log_trainer(engine):\n        evaluator.run(val_loader)\n        global all_pred\n        global all_true\n        validation_acc = evaluator.state.metrics[\"accuracy\"]\n        amex_score = amex_metric(all_true, all_pred)\n        all_pred = []\n        all_true = []\n        score_list.append(amex_score)\n        print(\"Rate is {}\".format(optimizer.param_groups[0][\"lr\"]))\n        print(\"Epoch: {} val accuracy: {:.4f} | loss: {:.4f} | amex_score: {:.2f}\".format(engine.state.epoch, validation_acc, trainer.state.metrics[\"loss\"], amex_score))\n\n    @evaluator.on(Events.ITERATION_COMPLETED)\n    def output_collect(engine):\n        y_pred, y_true = evaluator.state.output\n        global all_pred\n        global all_true\n        all_pred += list(y_pred.cpu().softmax(1)[:, -1])\n        all_true += list(y_true.cpu())\n    trainer.run(train_loader, max_epochs=10)\n\n    torch.save(model.state_dict(), \"fold{}_score{}_{}.bin\".format(fold, score_list[-1], time.strftime(\"%m%d_%H%M\")))","metadata":{},"execution_count":null,"outputs":[]}]}