{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import logging\nfrom nltk import word_tokenize\nimport pandas as pd\nimport numpy as np\nimport gc\nimport os\nimport torch\nfrom torch import nn\nfrom torch.autograd import Variable\nimport torch.nn.functional as F\nfrom torch.nn.utils.rnn import pad_sequence\nfrom sklearn.metrics import f1_score\nfrom torch import optim\nimport torchtext\nimport random","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9c4f9fa90ab7195fd8bdee5e4172041e9954decd"},"cell_type":"code","source":"text = torchtext.data.Field(lower=True, batch_first=True, tokenize=word_tokenize)\nqid = torchtext.data.Field()\ntarget = torchtext.data.Field(sequential=False, use_vocab=False, is_target=True)\ntrain = torchtext.data.TabularDataset(path='../input/train.csv', format='csv',fields={'question_text': ('text',text),'target': ('target',target)})\ntest = torchtext.data.TabularDataset(path='../input/test.csv', format='csv',fields={'qid': ('qid', qid),'question_text': ('text', text)})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1f922716fd705a536f460103c2c5b144cf62a778"},"cell_type":"code","source":"print(os.listdir(\"../input/embeddings/wiki-news-300d-1M\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3f3930a4c7bb4cafd51733f69db89ad8cde57a9e"},"cell_type":"code","source":"text.build_vocab(train, test, min_freq=5)\nqid.build_vocab(test)\ntext.vocab.load_vectors(torchtext.vocab.Vectors('../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'))\nprint(text.vocab.vectors.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f438d492583aa5e0af1d1bb45d0518d40bd273bc"},"cell_type":"code","source":"random.seed(1215)\ntrain, val = train.split(split_ratio=0.9, random_state=random.getstate())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6340ab0935056b99221f8f8c84c0a4af80642087"},"cell_type":"code","source":"class BiLSTM(nn.Module):\n    def __init__(self, pretrained_lm, padding_idx, static=True, hidden_dim=128, lstm_layer=2, dropout=0.2):\n        \n        super(BiLSTM, self).__init__()\n        \n        self.hidden_dim = hidden_dim\n        self.dropout = nn.Dropout(p=dropout)\n        self.embedding = nn.Embedding.from_pretrained(pretrained_lm)\n        self.embedding.padding_idx = padding_idx\n        \n        if static:\n            self.embedding.weight.requires_grad = False\n        \n        self.lstm = nn.LSTM(input_size=self.embedding.embedding_dim,\n                            hidden_size=hidden_dim,\n                            num_layers=lstm_layer, \n                            dropout = dropout,\n                            bidirectional=True)\n\n        self.hidden2label = nn.Linear(hidden_dim*lstm_layer*2, 1)\n    \n    def forward(self, sents):\n        \n        x = self.embedding(sents)\n        x = torch.transpose(x, dim0=1, dim1=0)\n        \n        lstm_out, (h_n, c_n) = self.lstm(x)\n        \n        y = self.hidden2label(self.dropout(torch.cat([c_n[i,:, :] for i in range(c_n.shape[0])], dim=1)))\n        \n        return y","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0ce926972063df18050b3db613a0fb25a09a4684"},"cell_type":"code","source":"def training(epoch, model, eval_every, loss_func, optimizer, train_iter, val_iter, early_stop=1, warmup_epoch = 2):\n    \n    step = 0\n    max_loss = 1e5\n    no_improve_epoch = 0\n    no_improve_in_previous_epoch = False\n    fine_tuning = False\n    train_record = []\n    val_record = []\n    losses = []\n    \n    for e in range(epoch):\n        \n        if e >= warmup_epoch:\n            if no_improve_in_previous_epoch:\n                no_improve_epoch += 1\n                if no_improve_epoch >= early_stop:\n                    break\n            else:\n                no_improve_epoch = 0\n            no_improve_in_previous_epoch = True\n\n        if not fine_tuning and e >= warmup_epoch:\n            \n            model.embedding.weight.requires_grad = True\n            fine_tuning = True\n        \n        train_iter.init_epoch()\n        \n        for train_batch in iter(train_iter):\n            step += 1\n            model.train()\n            \n            x = train_batch.text.cuda()\n            y = train_batch.target.type(torch.Tensor).cuda()\n            \n            model.zero_grad()\n            \n            pred = model.forward(x).view(-1)\n            \n            loss = loss_function(pred, y)\n            losses.append(loss.cpu().data.numpy())\n            train_record.append(loss.cpu().data.numpy())\n            \n            loss.backward()\n            optimizer.step()\n            \n            if step % eval_every == 0:\n                \n                model.eval()\n                model.zero_grad()\n                val_loss = []\n                \n                for val_batch in iter(val_iter):\n                    val_x = val_batch.text.cuda()\n                    val_y = val_batch.target.type(torch.Tensor).cuda()\n                    val_pred = model.forward(val_x).view(-1)\n                    val_loss.append(loss_function(val_pred, val_y).cpu().data.numpy())\n                \n                val_record.append({'step': step, 'loss': np.mean(val_loss)})\n                \n                print('epcoh {:02} - step {:06} - train_loss {:.4f} - val_loss {:.4f} '.format(\n                            e, step, np.mean(losses), val_record[-1]['loss']))\n                \n                if e >= warmup_epoch:\n                    \n                    if val_record[-1]['loss'] <= max_loss:\n                        save(m=model, info={'step': step, 'epoch': e, 'train_loss': np.mean(losses),\n                                            'val_loss': val_record[-1]['loss']})\n                        max_loss = val_record[-1]['loss']\n                        no_improve_in_previous_epoch = False\n    \n\ndef save(m, info):\n    \n    torch.save(info, 'best_model.info')\n    torch.save(m, 'best_model.m')\n    \ndef load():\n    \n    m = torch.load('best_model.m')\n    info = torch.load('best_model.info')\n    \n    return m, info","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"91a05c1c1614dc402686523cc5e67ed07dc63adc"},"cell_type":"code","source":"batch_size = 128\n\ntrain_iter = torchtext.data.BucketIterator(dataset=train,\n                                               batch_size=batch_size,\n                                               sort_key=lambda x: x.text.__len__(),\n                                               shuffle=True,\n                                               sort=False)\n\nval_iter = torchtext.data.BucketIterator(dataset=val,\n                                             batch_size=batch_size,\n                                             sort_key=lambda x: x.text.__len__(),\n                                             train=False,\n                                             sort=False)\n\nmodel = BiLSTM(text.vocab.vectors, lstm_layer=2, padding_idx=text.vocab.stoi[text.pad_token], hidden_dim=128).cuda()\n\n# loss_function = nn.BCEWithLogitsLoss(pos_weight=torch.Tensor([pos_w]).cuda())\n\nloss_function = nn.BCEWithLogitsLoss()\n\noptimizer = optim.Adam(filter(lambda p: p.requires_grad, model.parameters()),\n                    lr=1e-3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"48634d775001f12c7f7d15a4a39d8ed135081a44"},"cell_type":"code","source":"training(model=model, epoch=20, eval_every=500,\n         loss_func=loss_function, optimizer=optimizer, train_iter=train_iter,\n        val_iter=val_iter, warmup_epoch=3, early_stop=2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1d5c3623660e83143bb4ebc63920fc2dbd27657b"},"cell_type":"code","source":"model, m_info = load()\nm_info","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d1078460cf1ecec595e2098b702389c4dcbf7541"},"cell_type":"code","source":"model.lstm.flatten_parameters()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2f8e635b51d22fe3a5740d291b25d45c7f7b0c70"},"cell_type":"code","source":"model.eval()\nval_pred = []\nval_true = []\nval_iter.init_epoch()\nfor val_batch in iter(val_iter):\n    val_x = val_batch.text.cuda()\n    val_true += val_batch.target.data.numpy().tolist()\n    val_pred += torch.sigmoid(model.forward(val_x).view(-1)).cpu().data.numpy().tolist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8751a7ec41d599fc8e86d38b483cc6a8a1e47a3f"},"cell_type":"code","source":"tmp = [0,0,0] # idx, cur, max\ndelta = 0\nfor tmp[0] in np.arange(0.1, 0.501, 0.01):\n    tmp[1] = f1_score(val_true, np.array(val_pred)>tmp[0])\n    if tmp[1] > tmp[2]:\n        delta = tmp[0]\n        tmp[2] = tmp[1]\n\nprint('best threshold is {:.4f} with F1 score: {:.4f}'.format(delta, tmp[2]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5ede74ac7b63100716ea60ce7e0cb725f93a7038"},"cell_type":"code","source":"model.eval()\nmodel.zero_grad()\ntest_iter = torchtext.data.BucketIterator(dataset=test,\n                                    batch_size=batch_size,\n                                    sort_key=lambda x: x.text.__len__(),\n                                    sort=True)\ntest_pred = []\ntest_id = []\n\nfor test_batch in iter(test_iter):\n    test_x = test_batch.text.cuda()\n    test_pred += torch.sigmoid(model.forward(test_x).view(-1)).cpu().data.numpy().tolist()\n    test_id += test_batch.qid.view(-1).data.numpy().tolist()\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e9e288139f20456c1c72adeca5e540665a398e42"},"cell_type":"code","source":"sub_df =pd.DataFrame()\nsub_df['qid'] = [qid.vocab.itos[i] for i in test_id]\nsub_df['prediction'] = (np.array(test_pred) >= delta).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6a1db67d218de0814b466e0865a98d1d4adf38fb"},"cell_type":"code","source":"sub_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}