{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# Libraries installed are defined by the docker image: https://github.com/kaggle/docker-python\n\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torchtext\nimport random\nimport pdb\nfrom time import time\nfrom torch import nn\nfrom sklearn.metrics import f1_score\nfrom nltk import word_tokenize\nfrom torch import optim\n\nimport os\nprint(os.listdir(\"../input\"))\n\nrandom.seed(420)\nrandom_state = random.getstate()\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"602044bc637b1508acda0db98d3d1030ac82cb09"},"cell_type":"code","source":"text = torchtext.data.Field(lower=True, batch_first=True, tokenize=word_tokenize)\ntarget = torchtext.data.Field(sequential=False, use_vocab=False, is_target=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4e4f4f7791736612bf4562ee9e1c4d64a74fd30a"},"cell_type":"code","source":"data = torchtext.data.TabularDataset(path='../input/train.csv', format='csv',\n                                      fields={'question_text': ('text',text),\n                                              'target': ('target',target)})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f79416484ef09237a650b3955553b4d7d98eab27"},"cell_type":"code","source":"text.build_vocab(data, min_freq=1) # Tried 2. 1 is better\ntext.vocab.load_vectors(torchtext.vocab.Vectors('../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'))\nprint(text.vocab.vectors.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"824c525e8c73bf1fe3b9465261aa43e7251d815a"},"cell_type":"code","source":"train, validation = data.split(split_ratio=0.9, random_state=random_state)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"617bc464cf8a5c9d1ab4a601cfe77585109f9aaf"},"cell_type":"code","source":"batch_size = 64\n\ntrain_iter = torchtext.data.BucketIterator(dataset=train,\n                                           batch_size=batch_size,\n                                           sort_key=lambda x: x.text.__len__(),\n                                           shuffle=True,\n                                           sort=False)\n\nvalid_iter = torchtext.data.BucketIterator(dataset=validation,\n                                           batch_size=batch_size,\n                                           sort_key=lambda x: x.text.__len__(),\n                                           train=False,\n                                           sort=False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fef43f7960e472f00c0f29ded1be23d72732f0ee"},"cell_type":"markdown","source":"### Functions used during training and evaluation"},{"metadata":{"trusted":true,"_uuid":"37a2c07a7bf88b28199af5bb767164f97015be4d"},"cell_type":"code","source":"# Helper functions\ndef with_time(func):\n    pre = time()\n    res = func()\n    print(\"Elapsed time:\", time() - pre)\n    return res\n\ndef x_y_from_batch(batch):\n    x = batch.text.cuda()\n    y = batch.target.type(torch.Tensor).cuda()\n    \n    return x, y","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"defe0f8a7d4badf8eca48c7b2ae23da95ef618cb"},"cell_type":"code","source":"def training(\n    model, epoch, loss_func, optimizer, \n    no_improve_max=1, train_iter=train_iter, valid_iter=valid_iter\n):\n    step = 0\n    no_improve_streak = 0\n    val_record = []\n\n    for e in range(epoch):\n        train_record = []\n        train_iter.init_epoch()\n        \n        for train_batch in iter(train_iter):\n            step += 1\n            \n            model.train().zero_grad()\n            \n            x, y = x_y_from_batch(train_batch)\n            \n            pred = model.forward(x).view(-1)\n            \n            loss = loss_function(pred, y)\n            loss_data = loss.cpu().data.numpy()\n            train_record.append(loss_data)\n            loss.backward()\n            optimizer.step()\n            \n            if step % 1000 == 0:\n                print(\"Step {}, t loss {:.4f}\".format(step, loss_data))\n            \n        # end of epoch\n        model.eval().zero_grad()\n                \n        val_loss = []\n\n        for val_batch in iter(valid_iter):\n            val_x, val_y = x_y_from_batch(val_batch)\n\n            val_pred = model.forward(val_x).view(-1)\n            val_loss.append(loss_function(val_pred, val_y).cpu().data.numpy())\n\n        val_record.append({'step': step, 'loss': np.mean(val_loss)})\n\n        print('Epoch {} - step {} - train loss {:.4f} - valid loss {:.4f}'\n              .format(e, step, np.mean(train_record), val_record[-1]['loss'])\n         )\n        \n        # check for not improving\n        if e > 0 and (val_record[-1]['loss'] >= val_record[-2]['loss']):\n            no_improve_streak += 1\n            \n            if (no_improve_streak >= no_improve_max):\n                print('Reached no improve max!')\n                break\n        else:\n            no_improve_streak = 0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4fb194239f602b3f30a972f6fa8675dc3224681a"},"cell_type":"code","source":"class Model(nn.Module):\n    def __init__(self, pretrained_lm, padding_idx, lstm_size=128, lstm_layers=1, \n                 lstm_dropout=0, dropout1=0, \n                 with_hidden=False, dropout2=0, static=True):\n        super(Model, self).__init__()\n        \n        self.with_hidden = with_hidden\n        \n        self.embedding = nn.Embedding.from_pretrained(pretrained_lm)\n        self.embedding.padding_idx = padding_idx\n        \n        if static:\n            self.embedding.weight.requires_grad = False\n            \n        self.lstm = nn.LSTM(input_size=self.embedding.embedding_dim,\n                            hidden_size=lstm_size,\n                            num_layers=lstm_layers,\n                            dropout=lstm_dropout)\n        \n        self.dropout1 = nn.Dropout(p=dropout1)\n\n        hidden_size = lstm_size * lstm_layers\n            \n        if self.with_hidden:\n            hidden_size = lstm_size // 2\n            \n            self.hidden = nn.Linear(lstm_size * lstm_layers, hidden_size)\n            self.hidden_activ = nn.Tanh()\n            \n            self.dropout2 = nn.Dropout(p=dropout2)\n            \n        self.final = nn.Linear(hidden_size, 1)\n        \n    def apply_lstm(self, x):\n        _, (h_n, c_n) = self.lstm(x)\n        # gather all the hidden states\n        lstm_ = torch.cat([c_n[i, :, :] for i in range(c_n.shape[0])], dim=1)\n        \n        return self.dropout1(lstm_)\n        \n    def apply_hidden(self, x):\n        hidden = self.hidden(x)\n        hidden_act = self.hidden_activ(hidden)\n        \n        return self.dropout2(hidden_act)\n    \n    def forward(self, sents):\n        x = torch.transpose(self.embedding(sents), dim0=1, dim1=0)\n\n        hidden_res = self.apply_lstm(x)\n        \n        if self.with_hidden:\n            hidden_res = self.apply_hidden(hidden_res)\n        \n        output = self.final(hidden_res)\n        #output = self.final_activ(output)\n\n        return output","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"74874cd3ac052185a97f99e24545aa22cb4af0a8"},"cell_type":"code","source":"loss_function = nn.BCEWithLogitsLoss()\n\ndef get_optimizer(model, lr=0.001):\n    return optim.Adam(filter(lambda p: p.requires_grad, model.parameters()), lr=lr)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2b398b3e2e25b29c621e616eedca125c5e4c1533"},"cell_type":"code","source":"def model_score(model):\n    print('Preparing for evaluation...')\n\n    model.eval()\n    valid_iter.init_epoch()\n    \n    val_pred = []\n    val_true = []\n\n    for val_batch in iter(valid_iter):\n        val_true += val_batch.target.data.numpy().tolist()\n        val_pred += torch.sigmoid(\n                        model.forward(val_batch.text.cuda()).view(-1)\n                    ).cpu().data.numpy().tolist()\n\n    print('Evaluation started...')\n    \n    max_f1 = 0\n    thresh = 0\n\n    # Computing the best threshold based on f1 score\n    for thr in np.arange(0.1, 0.801, 0.01):\n        curr = f1_score(val_true, np.array(val_pred) > thr)\n\n        if curr > max_f1:\n            thresh = thr\n            max_f1 = curr\n\n    print('Best threshold is {:.3f} with F1 score: {:.4f}'.format(thresh, max_f1))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3084e6deba98fced7c2aeab1ec3c107bb40d1538"},"cell_type":"markdown","source":"### Training and evaluation"},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"ba6afe20119dbcf9606f874621290c0a67ce318c"},"cell_type":"code","source":"# This is the best model so far. Other models results will be shown during the presentation\nmodel = Model(\n    pretrained_lm=text.vocab.vectors,\n    padding_idx=text.vocab.stoi[text.pad_token],\n    lstm_size=128,\n    lstm_layers=1,\n    dropout1=0.5,\n    with_hidden=True,\n    dropout2=0.5 # hidden layer dropout\n).cuda()\n\nwith_time(lambda:\n    training(\n         model=model,\n         epoch=10, # should be big, it will automatically stop when starting to overfit\n         loss_func=loss_function,\n         optimizer=get_optimizer(model, 0.001)\n    )\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c16899aad9a80725d919222273d52ef13f48a4ff"},"cell_type":"code","source":"model_score(model)\n# Should be close to:\n# Best threshold is 0.450 with F1 score: 0.6844","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-output":false,"_kg_hide-input":false,"trusted":true,"_uuid":"b9a9e169a3e904337118a095306ab2d6251cd24e"},"cell_type":"markdown","source":"### NN summary:\nTesting of models with different inputs showed that:\n1. Minimal frequency of word during building vocabulary should be 1.\n2. There is no real benefit of using 2-layer LSTM.\n3. Dropout helps and should be approx 0.5.\n4. Sigmoid and Tanh showed almost the same results (Tanh a bit better).\n5. Idea of using two hidden layers after LSTM is bad.\n6. Two tested LSTM sizes (128 and 100) showed no significant difference.\n"},{"metadata":{"_uuid":"49331a4a0a4638012f195c04c48ae2866c05f6c7"},"cell_type":"markdown","source":"## Logistic regression"},{"metadata":{"trusted":true,"_uuid":"a569b837887b95b51fde0a9d32513f1518f16b80"},"cell_type":"code","source":"from plotly import tools\nimport plotly.offline as py\npy.init_notebook_mode(connected=True)\nimport plotly.graph_objs as go\n\nfrom sklearn import model_selection, preprocessing, metrics, ensemble, naive_bayes, linear_model\nfrom sklearn.feature_extraction.text import TfidfVectorizer, CountVectorizer\nfrom sklearn.decomposition import TruncatedSVD","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1d7db51d8f62ebbdea3c80653b293f3cbc44b89c"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c56e2c5acf0f0d8c2a283c4c31bc19c13c65aad1"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"af5f0661171b7e42d366eaa5f337587933ca8ef0"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"acbd239756e0d1ca69ca93ccd45a87032113e945"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}},"nbformat":4,"nbformat_minor":1}