{"cells":[{"metadata":{"trusted":true,"_uuid":"692039bf48f75d230f34ee89cb81512da02334ad","_kg_hide-output":false,"_kg_hide-input":false},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport torch\nimport torchtext\nimport random\nfrom torch import nn\nfrom sklearn.metrics import f1_score\nfrom nltk import word_tokenize\nfrom torch import optim\nfrom tqdm import tqdm\ntqdm.pandas()\n\ntrain = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")\nprint(\"Train shape : \",train.shape)\nprint(\"Test shape : \",test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a0fd85a6c8361e4965848ee4d4722a5339efe9b4"},"cell_type":"code","source":"def build_vocab(sentences, verbose =  True):\n    \"\"\"\n    :param sentences: list of list of words\n    :return: dictionary of words and their count\n    \"\"\"\n    vocab = {}\n    for sentence in tqdm(sentences, disable = (not verbose)):\n        for word in sentence:\n            try:\n                vocab[word] += 1\n            except KeyError:\n                vocab[word] = 1\n    return vocab","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9dd5b689a1f1b581c21f9559580ee7d25e3908d6"},"cell_type":"code","source":"sentences = train[\"question_text\"].progress_apply(lambda x: x.split()).values\nvocab = build_vocab(sentences)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aff88631cfa253d9e4b742c19a997a6e2abd11ac","scrolled":true},"cell_type":"code","source":"from gensim.models import KeyedVectors\n\nnews_path = '../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin'\nembeddings_index = KeyedVectors.load_word2vec_format(news_path, binary=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e9b5b3a6de3a798fcebb056ea99b1d1831609c3d"},"cell_type":"code","source":"import operator \n\ndef check_coverage(vocab,embeddings_index):\n    a = {}\n    oov = {}\n    k = 0\n    i = 0\n    for word in tqdm(vocab):\n        try:\n            a[word] = embeddings_index[word]\n            k += vocab[word]\n        except:\n\n            oov[word] = vocab[word]\n            i += vocab[word]\n            pass\n\n    print('Found embeddings for {:.2%} of vocab'.format(len(a) / len(vocab)))\n    print('Found embeddings for  {:.2%} of all text'.format(k / (k + i)))\n    sorted_x = sorted(oov.items(), key=operator.itemgetter(1))[::-1]\n\n    return sorted_x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7a2ad4950e60a9af67bdaf696039252d9bb594f7"},"cell_type":"code","source":"def clean_text(x):\n\n    x = str(x)\n    for punct in \"/-'\":\n        x = x.replace(punct, ' ')\n    for punct in '&':\n        x = x.replace(punct, f' {punct} ')\n    for punct in '?!.,\"#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~' + '“”’':\n        x = x.replace(punct, '')\n    return x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9a6d7d7dc8774de8a360ac3c99b439c20a5955d2"},"cell_type":"code","source":"train[\"question_text\"] = train[\"question_text\"].progress_apply(lambda x: clean_text(x))\nsentences = train[\"question_text\"].apply(lambda x: x.split())\nvocab = build_vocab(sentences)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fdd1f7a5bef010696c67cdb13061e263356f1a66"},"cell_type":"code","source":"import re\n\ndef clean_numbers(x):\n\n    x = re.sub('[0-9]{5,}', '#####', x)\n    x = re.sub('[0-9]{4}', '####', x)\n    x = re.sub('[0-9]{3}', '###', x)\n    x = re.sub('[0-9]{2}', '##', x)\n    return x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1b78e62c20013a1f76570d30debf1d0fc774d8c2"},"cell_type":"code","source":"train[\"question_text\"] = train[\"question_text\"].progress_apply(lambda x: clean_numbers(x))\nsentences = train[\"question_text\"].progress_apply(lambda x: x.split())\nvocab = build_vocab(sentences)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"482d469e793dad2bc1b548e823813ba80dfd2c5e"},"cell_type":"code","source":"def _get_mispell(mispell_dict):\n    mispell_re = re.compile('(%s)' % '|'.join(mispell_dict.keys()))\n    return mispell_dict, mispell_re\n\n\nmispell_dict = {'colour':'color',\n                'centre':'center',\n                'didnt':'did not',\n                'doesnt':'does not',\n                'isnt':'is not',\n                'shouldnt':'should not',\n                'favourite':'favorite',\n                'travelling':'traveling',\n                'counselling':'counseling',\n                'theatre':'theater',\n                'cancelled':'canceled',\n                'labour':'labor',\n                'organisation':'organization',\n                'wwii':'world war 2',\n                'citicise':'criticize',\n                'instagram': 'social medium',\n                'whatsapp': 'social medium',\n                'snapchat': 'social medium'\n\n                }\nmispellings, mispellings_re = _get_mispell(mispell_dict)\n\ndef replace_typical_misspell(text):\n    def replace(match):\n        return mispellings[match.group(0)]\n\n    return mispellings_re.sub(replace, text)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"17661ff2b23fa58d3dee2c956521461cfb10180c"},"cell_type":"code","source":"train[\"question_text\"] = train[\"question_text\"].progress_apply(lambda x: replace_typical_misspell(x))\nsentences = train[\"question_text\"].progress_apply(lambda x: x.split())\nto_remove = ['a','to','of','and']\nsentences = [[word for word in sentence if not word in to_remove] for sentence in tqdm(sentences)]\nvocab = build_vocab(sentences)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"377b803b6f9feb87f4b2ac3d01c9c6fb8521cacc"},"cell_type":"code","source":"oov = check_coverage(vocab,embeddings_index)\noov[:20]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dc93751f19dc28b57ff4bb3ee7b402bdd9f5a515"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nrandom_state = 43\nbatch_size = 64\n\ntrain, test = train_test_split(train, test_size=0.2, random_state=43)\n\n\ntrain_iter = torchtext.data.BucketIterator(dataset=train,\n                                           batch_size=batch_size,\n                                           sort_key=lambda x: x.text.__len__(),\n                                           shuffle=True,\n                                           sort=False)\n\nval_iter = torchtext.data.BucketIterator(dataset=val,\n                                         batch_size=batch_size,\n                                         sort_key=lambda x: x.text.__len__(),\n                                         train=False,\n                                         sort=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ace1aeef89e589c487f057b1f1d0ae2a22c287e6"},"cell_type":"code","source":"print(vocab)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"23916aeccaaa64894aa131f5b03083585b8e8e2f"},"cell_type":"code","source":"def training(epoch, model, loss_func, optimizer, train_iter, val_iter):\n    step = 0\n    train_record = []\n    losses = []\n    val_record = []\n    \n    for e in range(epoch):\n        train_iter.init_epoch()\n        for train_batch in iter(train_iter):\n            step += 1\n            model.train()\n            x = train_batch.text.cuda()\n            y = train_batch.target.type(torch.Tensor).cuda()\n            model.zero_grad()\n            pred = model.forward(x).view(-1)\n            loss = loss_function(pred, y)\n            loss_data = loss.cpu().data.numpy()\n            train_record.append(loss_data)\n            loss.backward()\n            optimizer.step()\n            if step % 1000 == 0:\n                print(\"Step: {:06}, loss {:.4f}\".format(step, loss_data))\n            if step % 10000 == 0:\n                model.eval()\n                model.zero_grad()\n                val_loss = []\n                for val_batch in iter(val_iter):\n                    val_x = val_batch.text.cuda()\n                    val_y = val_batch.target.type(torch.Tensor).cuda()\n                    val_pred = model.forward(val_x).view(-1)\n                    val_loss.append(loss_function(val_pred, val_y).cpu().data.numpy())\n                val_record.append({'step': step, 'loss': np.mean(val_loss)})\n                print('Epoch x{:02} - step {:06} - train_loss {:.4f} - val_loss {:.4f} '.format(\n                            e, step, np.mean(train_record), val_record[-1]['loss']))\n                train_record = []","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2f90006115ad364e3492afa66176a2de1209453d"},"cell_type":"code","source":"class SimpleModel(nn.Module):\n    def __init__(self, pretrained_lm, padding_idx, hidden_dim, static=True):\n        super(SimpleModel, self).__init__()\n        self.hidden_dim = 32\n        self.embedding = nn.Embedding.from_pretrained(pretrained_lm)\n        self.embedding.padding_idx = padding_idx\n        if static:\n            self.embedding.weight.requires_grad = False\n        self.lstm = nn.LSTM(input_size=self.embedding.embedding_dim,\n                            hidden_size=self.hidden_dim,\n                            num_layers=1)\n        self.lstm_to_linear = nn.Linear(self.hidden_dim, 1)\n        \n    def forward(self, sents):\n        x = self.embedding(sents)\n        x = torch.transpose(x, dim0=1, dim1=0)\n        lstm_out, (h_n, c_n) = self.lstm(x)\n        # gather all the hidden states\n        x = torch.cat([c_n[i,:, :] for i in range(c_n.shape[0])], dim=1)\n        # feed linear layer\n        output = self.lstm_to_linear(x)\n        return output","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d9e5bb7cbbf24024dcd7117ce67a0dc229f7ab98"},"cell_type":"code","source":"model = SimpleModel(vocab,\n                    padding_idx=text.vocab.stoi['<pad>'],\n                    hidden_dim=128).cuda()\nloss_function = nn.BCEWithLogitsLoss()\noptimizer = optim.Adam(filter(lambda p: p.requires_grad, model.parameters()),lr=1e-3)\n\ntraining(model=model,\n         epoch=20,\n         loss_func=loss_function,\n         optimizer=optimizer,\n         train_iter=train_iter,\n         val_iter=val_iter)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"84071f8ba9cc7c399e0775e66d77ecec28e9610b"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}