{"cells":[{"metadata":{"_uuid":"710ed17d0c57bd287be0ee3b2782a53a54510561"},"cell_type":"markdown","source":"## Summary\n\nThe idea of using a CNN to classify text was first presented in the paper [Convolutional Neural Networks for Sentence Classification](https://www.aclweb.org/anthology/D14-1181) by Yoon \nKim. In this Kernel I am trying to code up this network in Pytorch as well as Keras for later documentation as well as for learning purpose.\n\n<div style=\"margin-top: 9px; margin-bottom: 10px;\">\n<center><img src=\"https://mlwhiz.com/images/text_convolution.png\"  height=\"400\" width=\"700\" ></center>\n</div>\n\n\nCheck out my [blog post](https://mlwhiz.com/blog/2019/03/09/deeplearning_architectures_text_classification/) for more information\n"},{"metadata":{"_uuid":"97b92845b85f289ba795c8c8f7117526abe073d0"},"cell_type":"markdown","source":"## IMPORTS "},{"metadata":{"_uuid":"abb7e3c30b8a412a50c6b451c49939e3cf4bc11b","scrolled":true,"trusted":true},"cell_type":"code","source":"import random\nimport copy\nimport time\nimport pandas as pd\nimport numpy as np\nimport gc\nimport re\nimport torch\nfrom torchtext import data\n#import spacy\nfrom tqdm import tqdm_notebook, tnrange\nfrom tqdm.auto import tqdm\n\ntqdm.pandas(desc='Progress')\nfrom collections import Counter\nfrom textblob import TextBlob\nfrom nltk import word_tokenize\n\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.nn.utils.rnn import pack_padded_sequence, pad_packed_sequence\nfrom torch.autograd import Variable\nfrom torchtext.data import Example\nfrom sklearn.metrics import f1_score\nimport torchtext\nimport os \n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\n\n# cross validation and metrics\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import f1_score\nfrom torch.optim.optimizer import Optimizer\nfrom unidecode import unidecode\n\nfrom sklearn.preprocessing import StandardScaler\nfrom textblob import TextBlob\nfrom multiprocessing import  Pool\nfrom functools import partial\nimport numpy as np\nfrom sklearn.decomposition import PCA\nimport torch as t\nimport torch.nn as nn\nimport torch.nn.functional as F\n\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9a4ff5590a6f152dc1bec5aeca79aef10218f7de"},"cell_type":"markdown","source":"### Basic Parameters"},{"metadata":{"_uuid":"deee49df5ca1c4413f71677939e26aa1ff784e44","scrolled":true,"trusted":true},"cell_type":"code","source":"embed_size = 300 # how big is each word vector\nmax_features = 120000 # how many unique words to use (i.e num rows in embedding vector)\nmaxlen = 70 # max number of words in a question to use\nbatch_size = 512 # how many samples to process at once\nn_epochs = 5 # how many times to iterate over all samples\nn_splits = 5 # Number of K-fold Splits\nSEED = 10\ndebug = 0","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b53b0ebd37575ab31361099f0538cbb0457db5b6","trusted":true},"cell_type":"code","source":"loss_fn = torch.nn.BCEWithLogitsLoss(reduction='sum')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"654cbe3c8a1f2a618a2441afe00df3b4a89e0a58"},"cell_type":"markdown","source":"### Ensure determinism in the results\n\nA common headache in this competition is the lack of determinism in the results due to cudnn. The following Kernel has a solution in Pytorch.\n\nSee https://www.kaggle.com/hengzheng/pytorch-starter. "},{"metadata":{"_uuid":"58bbf87335799247586aaed16531f4d28d10ed4a","scrolled":true,"trusted":true},"cell_type":"code","source":"def seed_everything(seed=10):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\nseed_everything()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c890692644acce2dc4f6e2f929d6d294faca4ad2"},"cell_type":"markdown","source":"### Code for Loading Embeddings\n\nFunctions taken from the kernel:https://www.kaggle.com/gmhost/gru-capsule\n"},{"metadata":{"_uuid":"7026ee1d913f54dd4b560f654efdb9f833581cd3","scrolled":true,"trusted":true},"cell_type":"code","source":"## FUNCTIONS TAKEN FROM https://www.kaggle.com/gmhost/gru-capsule\n\ndef load_glove(word_index):\n    EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')[:300]\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE))\n    \n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = -0.005838499,0.48782197\n    embed_size = all_embs.shape[1]\n\n    # word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        #ALLmight\n        if embedding_vector is not None: \n            embedding_matrix[i] = embedding_vector\n        else:\n            embedding_vector = embeddings_index.get(word.capitalize())\n            if embedding_vector is not None: \n                embedding_matrix[i] = embedding_vector\n    return embedding_matrix \n    \n            \ndef load_fasttext(word_index):    \n    EMBEDDING_FILE = '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE) if len(o)>100)\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n\n    # word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n\n    return embedding_matrix\n\ndef load_para(word_index):\n    EMBEDDING_FILE = '../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt'\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE, encoding=\"utf8\", errors='ignore') if len(o)>100)\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = -0.0053247833,0.49346462\n    embed_size = all_embs.shape[1]\n\n    # word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n    \n    return embedding_matrix","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"07e9890ec0b490cef57565f7dff953aa56ebd3dc"},"cell_type":"markdown","source":"## Normalization\n\nBorrowed from:\n* How to: Preprocessing when using embeddings\nhttps://www.kaggle.com/christofhenkel/how-to-preprocessing-when-using-embeddings\n* Improve your Score with some Text Preprocessing https://www.kaggle.com/theoviel/improve-your-score-with-some-text-preprocessing"},{"metadata":{"_uuid":"28ebb28ba78972bb8d4fee9b53437045542d20fb","scrolled":true,"trusted":true},"cell_type":"code","source":"def build_vocab(texts):\n    sentences = texts.apply(lambda x: x.split()).values\n    vocab = {}\n    for sentence in sentences:\n        for word in sentence:\n            try:\n                vocab[word] += 1\n            except KeyError:\n                vocab[word] = 1\n    return vocab\n\ndef known_contractions(embed):\n    known = []\n    for contract in contraction_mapping:\n        if contract in embed:\n            known.append(contract)\n    return known\n\ndef clean_contractions(text, mapping):\n    specials = [\"’\", \"‘\", \"´\", \"`\"]\n    for s in specials:\n        text = text.replace(s, \"'\")\n    text = ' '.join([mapping[t] if t in mapping else t for t in text.split(\" \")])\n    return text\n\ndef correct_spelling(x, dic):\n    for word in dic.keys():\n        x = x.replace(word, dic[word])\n    return x\n\ndef unknown_punct(embed, punct):\n    unknown = ''\n    for p in punct:\n        if p not in embed:\n            unknown += p\n            unknown += ' '\n    return unknown\n\ndef clean_special_chars(text, punct, mapping):\n    for p in mapping:\n        text = text.replace(p, mapping[p])\n    \n    for p in punct:\n        text = text.replace(p, f' {p} ')\n    \n    specials = {'\\u200b': ' ', '…': ' ... ', '\\ufeff': '', 'करना': '', 'है': ''}  # Other special characters that I have to deal with in last\n    for s in specials:\n        text = text.replace(s, specials[s])\n    \n    return text\n\ndef add_lower(embedding, vocab):\n    count = 0\n    for word in vocab:\n        if word in embedding and word.lower() not in embedding:\n            embedding[word.lower()] = embedding[word]\n            count += 1\n    print(f\"Added {count} words to embedding\")    \n    ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"abeab4c80d6829cf2eae706bfa7929e2871af81f","scrolled":true,"trusted":true},"cell_type":"code","source":"puncts = [',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\', '•',  '~', '@', '£', \n '·', '_', '{', '}', '©', '^', '®', '`',  '<', '→', '°', '€', '™', '›',  '♥', '←', '×', '§', '″', '′', 'Â', '█', '½', 'à', '…', \n '“', '★', '”', '–', '●', 'â', '►', '−', '¢', '²', '¬', '░', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥', '▓', '—', '‹', '─', \n '▒', '：', '¼', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', 'é', '¯', '♦', '¤', '▲', 'è', '¸', '¾', 'Ã', '⋅', '‘', '∞', \n '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '³', '・', '╦', '╣', '╔', '╗', '▬', '❤', 'ï', 'Ø', '¹', '≤', '‡', '√', ]\n\ndef clean_text(x):\n    x = str(x)\n    for punct in puncts:\n        if punct in x:\n            x = x.replace(punct, f' {punct} ')\n    return x\n\n\ndef clean_numbers(x):\n    if bool(re.search(r'\\d', x)):\n        x = re.sub('[0-9]{5,}', '#####', x)\n        x = re.sub('[0-9]{4}', '####', x)\n        x = re.sub('[0-9]{3}', '###', x)\n        x = re.sub('[0-9]{2}', '##', x)\n    return x\n\nmispell_dict = {\"ain't\": \"is not\", \"aren't\": \"are not\",\"can't\": \"cannot\", \"'cause\": \"because\", \"could've\": \"could have\", \"couldn't\": \"could not\", \"didn't\": \"did not\",  \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\", \"he'd\": \"he would\",\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\", \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\",  \"I'd\": \"I would\", \"I'd've\": \"I would have\", \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"I've\": \"I have\", \"i'd\": \"i would\", \"i'd've\": \"i would have\", \"i'll\": \"i will\",  \"i'll've\": \"i will have\",\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'd've\": \"it would have\", \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\", \"ma'am\": \"madam\", \"mayn't\": \"may not\", \"might've\": \"might have\",\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \"must've\": \"must have\", \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \"needn't\": \"need not\", \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\", \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\", \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\", \"she'll\": \"she will\", \"she'll've\": \"she will have\", \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\", \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \"there'd've\": \"there would have\", \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\", \"they'll\": \"they will\", \"they'll've\": \"they will have\", \"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\", \"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\", \"we'll've\": \"we will have\", \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\", \"what're\": \"what are\",  \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\", \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\"you'd\": \"you would\", \"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\", 'colour': 'color', 'centre': 'center', 'favourite': 'favorite', 'travelling': 'traveling', 'counselling': 'counseling', 'theatre': 'theater', 'cancelled': 'canceled', 'labour': 'labor', 'organisation': 'organization', 'wwii': 'world war 2', 'citicise': 'criticize', 'youtu ': 'youtube ', 'Qoura': 'Quora', 'sallary': 'salary', 'Whta': 'What', 'narcisist': 'narcissist', 'howdo': 'how do', 'whatare': 'what are', 'howcan': 'how can', 'howmuch': 'how much', 'howmany': 'how many', 'whydo': 'why do', 'doI': 'do I', 'theBest': 'the best', 'howdoes': 'how does', 'mastrubation': 'masturbation', 'mastrubate': 'masturbate', \"mastrubating\": 'masturbating', 'pennis': 'penis', 'Etherium': 'Ethereum', 'narcissit': 'narcissist', 'bigdata': 'big data', '2k17': '2017', '2k18': '2018', 'qouta': 'quota', 'exboyfriend': 'ex boyfriend', 'airhostess': 'air hostess', \"whst\": 'what', 'watsapp': 'whatsapp', 'demonitisation': 'demonetization', 'demonitization': 'demonetization', 'demonetisation': 'demonetization'}\n\ndef _get_mispell(mispell_dict):\n    mispell_re = re.compile('(%s)' % '|'.join(mispell_dict.keys()))\n    return mispell_dict, mispell_re\n\nmispellings, mispellings_re = _get_mispell(mispell_dict)\ndef replace_typical_misspell(text):\n    def replace(match):\n        return mispellings[match.group(0)]\n    return mispellings_re.sub(replace, text)\n\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3c09d981ae674e6a373189a04dba8d0932b0765b"},"cell_type":"markdown","source":"Extra feature part taken from https://github.com/wongchunghang/toxic-comment-challenge-lstm/blob/master/toxic_comment_9872_model.ipynb"},{"metadata":{"_uuid":"63cb21525251b060aeb309e7be4b48772f8720f5","scrolled":true,"trusted":true},"cell_type":"code","source":"def load_and_prec():\n    if debug:\n        train_df = pd.read_csv(\"../input/train.csv\")[:80000]\n        test_df = pd.read_csv(\"../input/test.csv\")[:20000]\n    else:\n        train_df = pd.read_csv(\"../input/train.csv\")\n        test_df = pd.read_csv(\"../input/test.csv\")\n    print(\"Train shape : \",train_df.shape)\n    print(\"Test shape : \",test_df.shape)\n\n    \n    # lower the question text\n    train_df[\"question_text\"] = train_df[\"question_text\"].apply(lambda x: x.lower())\n    test_df[\"question_text\"] = test_df[\"question_text\"].apply(lambda x: x.lower())\n\n    # Clean the text\n    train_df[\"question_text\"] = train_df[\"question_text\"].progress_apply(lambda x: clean_text(x))\n    test_df[\"question_text\"] = test_df[\"question_text\"].apply(lambda x: clean_text(x))\n    \n    # Clean numbers\n    train_df[\"question_text\"] = train_df[\"question_text\"].progress_apply(lambda x: clean_numbers(x))\n    test_df[\"question_text\"] = test_df[\"question_text\"].apply(lambda x: clean_numbers(x))\n    \n    # Clean spellings\n    train_df[\"question_text\"] = train_df[\"question_text\"].progress_apply(lambda x: replace_typical_misspell(x))\n    test_df[\"question_text\"] = test_df[\"question_text\"].apply(lambda x: replace_typical_misspell(x))\n    \n    ## fill up the missing values\n    train_X = train_df[\"question_text\"].fillna(\"_##_\").values\n    test_X = test_df[\"question_text\"].fillna(\"_##_\").values\n\n    ## Tokenize the sentences\n    tokenizer = Tokenizer(num_words=max_features)\n    tokenizer.fit_on_texts(list(train_X))\n    train_X = tokenizer.texts_to_sequences(train_X)\n    test_X = tokenizer.texts_to_sequences(test_X)\n\n    ## Pad the sentences \n    train_X = pad_sequences(train_X, maxlen=maxlen)\n    test_X = pad_sequences(test_X, maxlen=maxlen)\n\n    ## Get the target values\n    train_y = train_df['target'].values\n    \n    #shuffling the data\n \n    np.random.seed(SEED)\n    trn_idx = np.random.permutation(len(train_X))\n\n    train_X = train_X[trn_idx]\n    train_y = train_y[trn_idx]\n    \n    \n    return train_X, test_X, train_y, tokenizer.word_index\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3c72fcddb4f680879e231c3dbfc0c71e27fc424c","scrolled":true,"trusted":true},"cell_type":"code","source":"start = time.time()\n\nx_train, x_test, y_train, word_index = load_and_prec() \n\nprint(time.time()-start)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e5c51a8329d569d13b9f0369ebb98ca8e2e55440"},"cell_type":"markdown","source":"### Load Embeddings\n\nChanging method: first load index and then load embedding\nTwo embedding matrices have been used. Glove, and paragram. The mean of the two is used as the final embedding matrix"},{"metadata":{"_uuid":"6a5f4502324d369ff6faa3692accee4f8a233005","scrolled":true,"trusted":true},"cell_type":"code","source":"# missing entries in the embedding are set using np.random.normal so we have to seed here too\nseed_everything()\nif debug:\n    paragram_embeddings = np.random.randn(120000,300)\n    glove_embeddings = np.random.randn(120000,300)\n    embedding_matrix = np.mean([glove_embeddings, paragram_embeddings], axis=0)\nelse:\n    glove_embeddings = load_glove(word_index)    \n    paragram_embeddings = load_para(word_index)\n    embedding_matrix = np.mean([glove_embeddings, paragram_embeddings], axis=0)        ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aa6a41607b804d76a2ddc530c912b5673bcd2423"},"cell_type":"code","source":"np.shape(embedding_matrix)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0a78496e4d88d8fb351cdf26d02f1554821ed445"},"cell_type":"markdown","source":"## Running Pytorch Model"},{"metadata":{"_uuid":"0da30e2afce23b753796f3045b44ce91a07e4303"},"cell_type":"markdown","source":"### Pytorch Run Functions"},{"metadata":{"trusted":true,"_uuid":"6a5afb54f70a29808af19946ba08ef971d194e46"},"cell_type":"code","source":"# code inspired from: https://github.com/anandsaha/pytorch.cyclic.learning.rate/blob/master/cls.py\nclass CyclicLR(object):\n    def __init__(self, optimizer, base_lr=1e-3, max_lr=6e-3,\n                 step_size=2000, mode='triangular', gamma=1.,\n                 scale_fn=None, scale_mode='cycle', last_batch_iteration=-1):\n\n        if not isinstance(optimizer, Optimizer):\n            raise TypeError('{} is not an Optimizer'.format(\n                type(optimizer).__name__))\n        self.optimizer = optimizer\n\n        if isinstance(base_lr, list) or isinstance(base_lr, tuple):\n            if len(base_lr) != len(optimizer.param_groups):\n                raise ValueError(\"expected {} base_lr, got {}\".format(\n                    len(optimizer.param_groups), len(base_lr)))\n            self.base_lrs = list(base_lr)\n        else:\n            self.base_lrs = [base_lr] * len(optimizer.param_groups)\n\n        if isinstance(max_lr, list) or isinstance(max_lr, tuple):\n            if len(max_lr) != len(optimizer.param_groups):\n                raise ValueError(\"expected {} max_lr, got {}\".format(\n                    len(optimizer.param_groups), len(max_lr)))\n            self.max_lrs = list(max_lr)\n        else:\n            self.max_lrs = [max_lr] * len(optimizer.param_groups)\n\n        self.step_size = step_size\n\n        if mode not in ['triangular', 'triangular2', 'exp_range'] \\\n                and scale_fn is None:\n            raise ValueError('mode is invalid and scale_fn is None')\n\n        self.mode = mode\n        self.gamma = gamma\n\n        if scale_fn is None:\n            if self.mode == 'triangular':\n                self.scale_fn = self._triangular_scale_fn\n                self.scale_mode = 'cycle'\n            elif self.mode == 'triangular2':\n                self.scale_fn = self._triangular2_scale_fn\n                self.scale_mode = 'cycle'\n            elif self.mode == 'exp_range':\n                self.scale_fn = self._exp_range_scale_fn\n                self.scale_mode = 'iterations'\n        else:\n            self.scale_fn = scale_fn\n            self.scale_mode = scale_mode\n\n        self.batch_step(last_batch_iteration + 1)\n        self.last_batch_iteration = last_batch_iteration\n\n    def batch_step(self, batch_iteration=None):\n        if batch_iteration is None:\n            batch_iteration = self.last_batch_iteration + 1\n        self.last_batch_iteration = batch_iteration\n        for param_group, lr in zip(self.optimizer.param_groups, self.get_lr()):\n            param_group['lr'] = lr\n\n    def _triangular_scale_fn(self, x):\n        return 1.\n\n    def _triangular2_scale_fn(self, x):\n        return 1 / (2. ** (x - 1))\n\n    def _exp_range_scale_fn(self, x):\n        return self.gamma**(x)\n\n    def get_lr(self):\n        step_size = float(self.step_size)\n        cycle = np.floor(1 + self.last_batch_iteration / (2 * step_size))\n        x = np.abs(self.last_batch_iteration / step_size - 2 * cycle + 1)\n\n        lrs = []\n        param_lrs = zip(self.optimizer.param_groups, self.base_lrs, self.max_lrs)\n        for param_group, base_lr, max_lr in param_lrs:\n            base_height = (max_lr - base_lr) * np.maximum(0, (1 - x))\n            if self.scale_mode == 'cycle':\n                lr = base_lr + base_height * self.scale_fn(cycle)\n            else:\n                lr = base_lr + base_height * self.scale_fn(self.last_batch_iteration)\n            lrs.append(lr)\n        return lrs\n    \nclass MyDataset(Dataset):\n    def __init__(self,dataset):\n        self.dataset = dataset\n    def __getitem__(self,index):\n        data,target = self.dataset[index]\n        return data,target,index\n    def __len__(self):\n        return len(self.dataset)\n\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b711f30cbb4b0297fec816b8102c477d9a546c80","trusted":true},"cell_type":"code","source":"def pytorch_model_run_cv(x_train,y_train,features,x_test, model_obj, feats = False,clip = True):\n    seed_everything()\n    avg_losses_f = []\n    avg_val_losses_f = []\n    # matrix for the out-of-fold predictions\n    train_preds = np.zeros((len(x_train)))\n    # matrix for the predictions on the test set\n    test_preds = np.zeros((len(x_test)))\n    splits = list(StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED).split(x_train, y_train))\n    for i, (train_idx, valid_idx) in enumerate(splits):\n        seed_everything(i*1000+i)\n        x_train = np.array(x_train)\n        y_train = np.array(y_train)\n        if feats:\n            features = np.array(features)\n        x_train_fold = torch.tensor(x_train[train_idx.astype(int)], dtype=torch.long).cuda()\n        y_train_fold = torch.tensor(y_train[train_idx.astype(int), np.newaxis], dtype=torch.float32).cuda()\n        if feats:\n            kfold_X_features = features[train_idx.astype(int)]\n            kfold_X_valid_features = features[valid_idx.astype(int)]\n        x_val_fold = torch.tensor(x_train[valid_idx.astype(int)], dtype=torch.long).cuda()\n        y_val_fold = torch.tensor(y_train[valid_idx.astype(int), np.newaxis], dtype=torch.float32).cuda()\n        \n        model = copy.deepcopy(model_obj)\n\n        model.cuda()\n\n        loss_fn = torch.nn.BCEWithLogitsLoss(reduction='sum')\n        optimizer = torch.optim.Adam(filter(lambda p: p.requires_grad, model.parameters()), \n                                 lr=0.001)\n        \n        ################################################################################################\n        scheduler = False\n        ###############################################################################################\n\n        train = MyDataset(torch.utils.data.TensorDataset(x_train_fold, y_train_fold))\n        valid = MyDataset(torch.utils.data.TensorDataset(x_val_fold, y_val_fold))\n        \n        train_loader = torch.utils.data.DataLoader(train, batch_size=batch_size, shuffle=True)\n        valid_loader = torch.utils.data.DataLoader(valid, batch_size=batch_size, shuffle=False)\n\n        print(f'Fold {i + 1}')\n        for epoch in range(n_epochs):\n            start_time = time.time()\n            model.train()\n\n            avg_loss = 0.  \n            for i, (x_batch, y_batch, index) in enumerate(train_loader):\n                if feats:       \n                    f = kfold_X_features[index]\n                    y_pred = model([x_batch,f])\n                else:\n                    y_pred = model(x_batch)\n\n                if scheduler:\n                    scheduler.batch_step()\n\n                # Compute and print loss.\n                loss = loss_fn(y_pred, y_batch)\n                optimizer.zero_grad()\n                loss.backward()\n                if clip:\n                    nn.utils.clip_grad_norm_(model.parameters(),1)\n                optimizer.step()\n                avg_loss += loss.item() / len(train_loader)\n                \n            model.eval()\n            \n            valid_preds_fold = np.zeros((x_val_fold.size(0)))\n            test_preds_fold = np.zeros((len(x_test)))\n            \n            avg_val_loss = 0.\n            for i, (x_batch, y_batch,index) in enumerate(valid_loader):\n                if feats:\n                    f = kfold_X_valid_features[index]            \n                    y_pred = model([x_batch,f]).detach()\n                else:\n                    y_pred = model(x_batch).detach()\n                \n                avg_val_loss += loss_fn(y_pred, y_batch).item() / len(valid_loader)\n                valid_preds_fold[index] = sigmoid(y_pred.cpu().numpy())[:, 0]\n            \n            elapsed_time = time.time() - start_time \n            print('Epoch {}/{} \\t loss={:.4f} \\t val_loss={:.4f} \\t time={:.2f}s'.format(\n                epoch + 1, n_epochs, avg_loss, avg_val_loss, elapsed_time))\n        avg_losses_f.append(avg_loss)\n        avg_val_losses_f.append(avg_val_loss) \n        # predict all samples in the test set batch per batch\n        for i, (x_batch,) in enumerate(test_loader):\n            if feats:\n                f = test_features[i * batch_size:(i+1) * batch_size]\n                y_pred = model([x_batch,f]).detach()\n            else:\n                y_pred = model(x_batch).detach()\n\n            test_preds_fold[i * batch_size:(i+1) * batch_size] = sigmoid(y_pred.cpu().numpy())[:, 0]\n            \n        train_preds[valid_idx] = valid_preds_fold\n        test_preds += test_preds_fold / len(splits)\n\n    print('All \\t loss={:.4f} \\t val_loss={:.4f} \\t '.format(np.average(avg_losses_f),np.average(avg_val_losses_f)))\n    return train_preds, test_preds","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b8caecb207a12d4a5524fe16e4524b31c7da8bac"},"cell_type":"markdown","source":"###  TextCNN Model in Pytorch\n"},{"metadata":{"_kg_hide-input":false,"_uuid":"26e90f1839b084c126414250eac1f636b0c88937","trusted":true},"cell_type":"code","source":"class CNN_Text(nn.Module):\n    \n    def __init__(self):\n        super(CNN_Text, self).__init__()\n        filter_sizes = [1,2,3,5]\n        num_filters = 36\n        self.embedding = nn.Embedding(max_features, embed_size)\n        self.embedding.weight = nn.Parameter(torch.tensor(embedding_matrix, dtype=torch.float32))\n        self.embedding.weight.requires_grad = False\n        self.convs1 = nn.ModuleList([nn.Conv2d(1, num_filters, (K, embed_size)) for K in filter_sizes])\n        self.dropout = nn.Dropout(0.1)\n        self.fc1 = nn.Linear(len(filter_sizes)*num_filters, 1)\n\n\n    def forward(self, x):\n        x = self.embedding(x)  \n        x = x.unsqueeze(1)  \n        x = [F.relu(conv(x)).squeeze(3) for conv in self.convs1] \n        x = [F.max_pool1d(i, i.size(2)).squeeze(2) for i in x]  \n        x = torch.cat(x, 1)\n        x = self.dropout(x)  \n        logit = self.fc1(x)  \n        return logit","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"551ff8837a1b0899e0925c1a3801547adeddccae","trusted":true},"cell_type":"code","source":"def sigmoid(x):\n    return 1 / (1 + np.exp(-x))\n\n# always call this before training for deterministic results\nseed_everything()\n\nx_test_cuda = torch.tensor(x_test, dtype=torch.long).cuda()\ntest = torch.utils.data.TensorDataset(x_test_cuda)\ntest_loader = torch.utils.data.DataLoader(test, batch_size=batch_size, shuffle=False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3b8f851d15f3e34d40eab4146393c145886b3966","trusted":true},"cell_type":"code","source":"train_preds , test_preds = pytorch_model_run_cv(x_train,y_train,None,x_test,CNN_Text(), feats = False, clip=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2ab6ac01d6d69d88ef05a542cb6f8e27cd23a8ab"},"cell_type":"code","source":"def bestThresshold(y_train,train_preds):\n    tmp = [0,0,0] # idx, cur, max\n    delta = 0\n    for tmp[0] in tqdm(np.arange(0.1, 0.501, 0.01)):\n        tmp[1] = f1_score(y_train, np.array(train_preds)>tmp[0])\n        if tmp[1] > tmp[2]:\n            delta = tmp[0]\n            tmp[2] = tmp[1]\n    print('best threshold is {:.4f} with F1 score: {:.4f}'.format(delta, tmp[2]))\n    return delta , tmp[2]\n\ndelta, _ = bestThresshold(y_train,train_preds)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"13dea8e1c900175db765f3fac41bcaec035d4476","trusted":true},"cell_type":"markdown","source":"## b. Runing Keras Model"},{"metadata":{"_uuid":"bf5c95169916bb4fa869eb7d14ac2b03377e486f"},"cell_type":"markdown","source":"### Keras Run Functions"},{"metadata":{"trusted":true,"_uuid":"70801ae169807ba2992348904a448c5d5490c599"},"cell_type":"code","source":"from keras.layers import Dense, Input, CuDNNLSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D, GlobalMaxPooling1D, GlobalAveragePooling1D\nfrom keras.layers import Input, Embedding, Dense, Conv2D, MaxPool2D, concatenate\nfrom keras.layers import Reshape, Flatten, Concatenate, Dropout, SpatialDropout1D\nfrom keras.optimizers import Adam\nfrom keras.models import Model\nfrom keras.engine.topology import Layer\nfrom keras import initializers, regularizers, constraints, optimizers, layers\n\n\nfrom keras.layers import *\nfrom keras.models import *\nfrom keras import initializers, regularizers, constraints, optimizers, layers\nfrom keras.initializers import *\nfrom keras.optimizers import *\nimport keras.backend as K\nfrom keras.callbacks import *\nimport tensorflow as tf\n\ndef model_train_cv(x_train,y_train,nfold,model_obj):\n    splits = list(StratifiedKFold(n_splits=nfold, shuffle=True, random_state=SEED).split(x_train, y_train))\n    x_train = x_train\n    y_train = np.array(y_train)\n    # matrix for the out-of-fold predictions\n    train_oof_preds = np.zeros((x_train.shape[0]))\n    for i, (train_idx, valid_idx) in enumerate(splits):\n        print(f'Fold {i + 1}')\n        x_train_fold = x_train[train_idx.astype(int)]\n        y_train_fold = y_train[train_idx.astype(int)]\n        x_val_fold = x_train[valid_idx.astype(int)]\n        y_val_fold = y_train[valid_idx.astype(int)]\n\n        clf = copy.deepcopy(model_obj)\n        clf.fit(x_train_fold, y_train_fold, batch_size=512, epochs=5, validation_data=(x_val_fold, y_val_fold))\n        \n        valid_preds_fold = clf.predict(x_val_fold)[:,0]\n\n        # storing OOF predictions\n        train_oof_preds[valid_idx] = valid_preds_fold\n    return train_oof_preds\n\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9506af4bf6bec66e40536a7ffa4df58b09aca100"},"cell_type":"markdown","source":"### TextCNN model in Keras"},{"metadata":{"trusted":true,"_uuid":"837f0f434c25cc95cf201e9a1ec06c1a735096e5"},"cell_type":"code","source":"# https://www.kaggle.com/yekenot/2dcnn-textclassifier\ndef model_cnn(embedding_matrix):\n    filter_sizes = [1,2,3,5]\n    num_filters = 36\n\n    inp = Input(shape=(maxlen,))\n    x = Embedding(max_features, embed_size, weights=[embedding_matrix],trainable=False)(inp)\n    x = Reshape((maxlen, embed_size, 1))(x)\n\n    maxpool_pool = []\n    for i in range(len(filter_sizes)):\n        conv = Conv2D(num_filters, kernel_size=(filter_sizes[i], embed_size),\n                                     kernel_initializer='he_normal', activation='relu')(x)\n        maxpool_pool.append(MaxPool2D(pool_size=(maxlen - filter_sizes[i] + 1, 1))(conv))\n\n    z = Concatenate(axis=1)(maxpool_pool)   \n    z = Flatten()(z)\n    z = Dropout(0.1)(z)\n\n    outp = Dense(1, activation=\"sigmoid\")(z)\n\n    model = Model(inputs=inp, outputs=outp)\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e0cb6f64029b11d84bb901289964ee5984933740"},"cell_type":"code","source":"model = model_cnn(embedding_matrix)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"daaa42f4b8d503682a10ba9b0f97373a24c6fb38"},"cell_type":"code","source":"train_oof_preds = model_train_cv(x_train,y_train,5,model)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"179a6501cb7a5bcfbf0f19c42767dbafdffdca72"},"cell_type":"code","source":"delta, _ = bestThresshold(y_train,train_oof_preds)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9d1b3fc0c8bc59f91a203adcf3dbd9d89759c8df","trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fc3e72f43d1ddba1a9d82be0cf9b642f6d5795d6","trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"10c39195707e4710e11d31bafd4bde1e21ff2d6f","trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"162772df446ca6ad6dda6a003d8ea2e58106192f","trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fdca46457c79c6b56409e6cfbc76d90de4d19cbe","trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"717f6f2fbaa37d01db018baafcf8216fee98ce6a","trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7a713658e272a3d7ec00267faa5e3d02362269ed","trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7ce9e3159ce70cdaa0d706249cf80ea3157b39d0","trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0fb26d857e36473fe7a458cf61692ceff5cadceb","trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8dee2aefa02222d4dd5e67640708cad59285a796","trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8682fc20ab59210e2aba1381232e68ce4e9519d2","trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2681fc633782d5a7005a0db95de14c6b05b88dac","trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"39e764384d4073282b6a51bcd1ecf382134a33a6","trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3703e549d2d8d5bb0948c2771b1e6636f0fccef2","trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ee777c692947883aaa2a04349e32c6f45d5702b5"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6e8cd0bbd6cfcd4dd507f4353c29b56cd7594fe2"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f6abad47bd7c3544fc8a3ba49438afa0831131dc"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b9f763bf089050aaa711fc1e371bf83fdca515e8"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"330cc55c14571e7a88586d88676c13ec2677efed"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"eca5b5595de60fd388a4ecb4f602c89eae83d1ec"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"af8a0a6e78675868655470b4b22a8352c15933bb"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"211b2441b513ff07c13c0a5f8920728235db9507"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bc490533a242766921044c4e672e0ae76808850b"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a7c69bba55bf136310a3b890404d03e1628d985b"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"428f9cad34c6eadaf54cf316d107c9d686441fbf"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"16c92267bbbc1d611b3f7f9a1dc1eea24e009b16"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}