{"cells":[{"metadata":{"_uuid":"38e7d605b52588dfa82fb54def70d25e511df5bd"},"cell_type":"markdown","source":"### Inspired by:\n* https://www.kaggle.com/sudalairajkumar/a-look-at-different-embeddings\n* https://www.kaggle.com/shujian/single-rnn-with-4-folds-v1-9\n* http://mlexplained.com/2018/01/13/weight-normalization-and-layer-normalization-explained-normalization-in-deep-learning-part-2/\n* https://arxiv.org/abs/1607.06450\n* https://github.com/keras-team/keras/issues/3878\n* https://www.kaggle.com/lystdo/lstm-with-word2vec-embeddings\n* https://www.kaggle.com/jhoward/improved-lstm-baseline-glove-dropout\n* https://www.kaggle.com/aquatic/entity-embedding-neural-net\n* https://www.kaggle.com/hireme/fun-api-keras-f1-metric-cyclical-learning-rate\n* https://ai.google/research/pubs/pub46697\n* https://blog.openai.com/quantifying-generalization-in-reinforcement-learning/\n* https://www.kaggle.com/rasvob/let-s-try-clr-v3\n* https://github.com/bentrevett/pytorch-sentiment-analysis/blob/master/3%20-%20Faster%20Sentiment%20Analysis.ipynb\n* https://www.kaggle.com/ziliwang/pytorch-text-cnn\n* https://github.com/yunjey/pytorch-tutorial/blob/master/tutorials/02-intermediate/bidirectional_recurrent_neural_network/main.py\n* https://github.com/clairett/pytorch-sentiment-classification/blob/master/bilstm.py\n* https://pytorch.org/tutorials/beginner/nlp/sequence_models_tutorial.html\n* https://www.kaggle.com/bminixhofer/a-validation-framework-impact-of-the-random-seed?scriptVersionId=9370229\n* https://www.kaggle.com/theoviel/improve-your-score-with-some-text-preprocessing\n\n\ntrying torch...\n\nmuch harder then keras, but feels more rewarding when done"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport sys\nnp.set_printoptions(threshold=sys.maxsize)\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\nprint(os.listdir(\"../input/embeddings\"))\nprint(os.listdir(\"../input/embeddings/GoogleNews-vectors-negative300\"))\n\n# Any results you write to the current directory are saved as output.\nimport torch\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report,f1_score,precision_recall_fscore_support,recall_score,precision_score\nfrom sklearn.utils import class_weight\nimport matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ebaa16183c3b53a318cc917737e7e1a88c15fb34"},"cell_type":"code","source":"import random\n\ndef seed_everything(seed=1234):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    \nseed_everything(6017)\nprint('Seeding done...')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bd86c03fdf44b58e9a817f8781bc567f279e1f3e"},"cell_type":"code","source":"df = pd.read_csv('../input/train.csv')\ndf[\"question_text\"].fillna(\"_##_\",inplace=True)\nmax_len = df['question_text'].apply(lambda x:len(x)).max()\nprint('max length of sequences:',max_len)\n# df = df.sample(frac=0.1)\n\nprint('columns:',df.columns)\npd.set_option('display.max_columns',None)\nprint('df head:',df.head())\nprint('example of the question text values:',df['question_text'].head().values)\nprint('what values contains target:',df.target.unique())\n\nprint('Loading test data...')\ndf_final = pd.read_csv('../input/test.csv')\ndf_final[\"question_text\"].fillna(\"_##_\", inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e0ea657f9468ed81bd9e7998454b473a0325ec11"},"cell_type":"code","source":"import re\n\nprint('Preproccesing texts....')\n\nprint('lower...')\ndf[\"question_text\"] = df[\"question_text\"].apply(lambda x: x.lower())\ndf_final[\"question_text\"] = df_final[\"question_text\"].apply(lambda x: x.lower())\n\ncontraction_mapping = {\n\"ain't\": \"is not\", \n\"aren't\": \"are not\",\n\"can't\": \"cannot\",\n \"'cause\": \"because\",\n \"could've\": \"could have\",\n \"couldn't\": \"could not\",\n \"didn't\": \"did not\",\n  \"doesn't\": \"does not\",\n \"don't\": \"do not\",\n \"hadn't\": \"had not\",\n \"hasn't\": \"has not\",\n \"haven't\": \"have not\",\n \"he'd\": \"he would\",\n\"he'll\": \"he will\",\n \"he's\": \"he is\",\n \"how'd\": \"how did\",\n \"how'd'y\": \"how do you\",\n \"how'll\": \"how will\",\n \"how's\": \"how is\",\n  \"I'd\": \"I would\",\n \"I'd've\": \"I would have\",\n \"I'll\": \"I will\",\n \"I'll've\": \"I will have\",\n\"I'm\": \"I am\",\n \"I've\": \"I have\",\n \"i'd\": \"i would\",\n \"i'd've\": \"i would have\",\n \"i'll\": \"i will\",\n  \"i'll've\": \"i will have\",\n\"i'm\": \"i am\",\n \"i've\": \"i have\",\n \"isn't\": \"is not\",\n \"it'd\": \"it would\",\n \"it'd've\": \"it would have\",\n \"it'll\": \"it will\",\n \"it'll've\": \"it will have\",\n\"it's\": \"it is\",\n \"let's\": \"let us\",\n \"ma'am\": \"madam\",\n \"mayn't\": \"may not\",\n \"might've\": \"might have\",\n\"mightn't\": \"might not\",\n\"mightn't've\": \"might not have\",\n \"must've\": \"must have\",\n \"mustn't\": \"must not\",\n \"mustn't've\": \"must not have\",\n \"needn't\": \"need not\",\n \"needn't've\": \"need not have\",\n\"o'clock\": \"of the clock\",\n \"oughtn't\": \"ought not\",\n \"oughtn't've\": \"ought not have\",\n \"shan't\": \"shall not\",\n \"sha'n't\": \"shall not\",\n \"shan't've\": \"shall not have\",\n \"she'd\": \"she would\",\n \"she'd've\": \"she would have\",\n \"she'll\": \"she will\",\n \"she'll've\": \"she will have\",\n \"she's\": \"she is\",\n \"should've\": \"should have\",\n \"shouldn't\": \"should not\",\n \"shouldn't've\": \"should not have\",\n \"so've\": \"so have\",\n\"so's\": \"so as\",\n \"this's\": \"this is\",\n\"that'd\": \"that would\",\n \"that'd've\": \"that would have\",\n \"that's\": \"that is\",\n \"there'd\": \"there would\",\n \"there'd've\": \"there would have\",\n \"there's\": \"there is\",\n \"here's\": \"here is\",\n\"they'd\": \"they would\",\n \"they'd've\": \"they would have\",\n \"they'll\": \"they will\",\n \"they'll've\": \"they will have\",\n \"they're\": \"they are\",\n \"they've\": \"they have\",\n \"to've\": \"to have\",\n \"wasn't\": \"was not\",\n \"we'd\": \"we would\",\n \"we'd've\": \"we would have\",\n \"we'll\": \"we will\",\n \"we'll've\": \"we will have\",\n \"we're\": \"we are\",\n \"we've\": \"we have\",\n \"weren't\": \"were not\",\n \"what'll\": \"what will\",\n \"what'll've\": \"what will have\",\n \"what're\": \"what are\",\n  \"what's\": \"what is\",\n \"what've\": \"what have\",\n \"when's\": \"when is\",\n \"when've\": \"when have\",\n \"where'd\": \"where did\",\n \"where's\": \"where is\",\n \"where've\": \"where have\",\n \"who'll\": \"who will\",\n \"who'll've\": \"who will have\",\n \"who's\": \"who is\",\n \"who've\": \"who have\",\n \"why's\": \"why is\",\n \"why've\": \"why have\",\n \"will've\": \"will have\",\n \"won't\": \"will not\",\n \"won't've\": \"will not have\",\n \"would've\": \"would have\",\n \"wouldn't\": \"would not\",\n \"wouldn't've\": \"would not have\",\n \"y'all\": \"you all\",\n \"y'all'd\": \"you all would\",\n\"y'all'd've\": \"you all would have\",\n\"y'all're\": \"you all are\",\n\"y'all've\": \"you all have\",\n\"you'd\": \"you would\",\n \"you'd've\": \"you would have\",\n \"you'll\": \"you will\",\n \"you'll've\": \"you will have\",\n \"you're\": \"you are\",\n \"you've\": \"you have\" }\n\ndef clean_contractions(text, mapping):\n    specials = [\"’\", \"‘\", \"´\", \"`\"]\n    for s in specials:\n        text = text.replace(s, \"'\")\n    text = ' '.join([mapping[t] if t in mapping else t for t in text.split(\" \")])\n    return text\n\nprint('contractions...')\ndf[\"question_text\"] = df[\"question_text\"].apply(lambda x: clean_contractions(x,contraction_mapping))\ndf_final[\"question_text\"] = df_final[\"question_text\"].apply(lambda x: clean_contractions(x,contraction_mapping))\n\n\npunct_mapping = {\"‘\": \"'\",\n \"₹\": \"e\",\n \"´\": \"'\",\n \"°\": \"\",\n \"€\": \"e\",\n \"™\": \"tm\",\n \"√\": \" sqrt \",\n \"×\": \"x\",\n \"²\": \"2\",\n \"—\": \"-\",\n \"–\": \"-\",\n \"’\": \"'\",\n \"_\": \"-\",\n \"`\": \"'\",\n '“': '\"',\n '”': '\"',\n '“': '\"',\n \"£\": \"e\",\n '∞': 'infinity',\n 'θ': 'theta',\n '÷': '/',\n 'α': 'alpha',\n '•': '.',\n 'à': 'a',\n '−': '-',\n 'β': 'beta',\n '∅': '',\n '³': '3',\n 'π': 'pi',\n }\n\npunct = [',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\', '•',  '~', '@', '£', \n '·', '_', '{', '}', '©', '^', '®', '`',  '<', '→', '°', '€', '™', '›',  '♥', '←', '×', '§', '″', '′', 'Â', '█', '½', 'à', '…', \n '“', '★', '”', '–', '●', 'â', '►', '−', '¢', '²', '¬', '░', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥', '▓', '—', '‹', '─', \n '▒', '：', '¼', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', 'é', '¯', '♦', '¤', '▲', 'è', '¸', '¾', 'Ã', '⋅', '‘', '∞', \n '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '³', '・', '╦', '╣', '╔', '╗', '▬', '❤', 'ï', 'Ø', '¹', '≤', '‡', '√', ]\n\ndef clean_special_chars(text, punct, mapping):\n    for p in mapping:\n        text = text.replace(p, mapping[p])\n    \n    for p in punct:\n        text = text.replace(p, f' {p} ')\n    \n    specials = {'\\u200b': ' ', '…': ' ... ', '\\ufeff': '', 'करना': '', 'है': ''}  # Other special characters that I have to deal with in last\n    for s in specials:\n        text = text.replace(s, specials[s])\n    \n    return text\n\nprint('clean special chars...')\ndf[\"question_text\"] = df[\"question_text\"].apply(lambda x: clean_special_chars(x, punct, punct_mapping))\ndf_final[\"question_text\"] = df_final[\"question_text\"].apply(lambda x: clean_special_chars(x, punct, punct_mapping))\n\nmispell_dict = {'colour': 'color',\n 'centre': 'center',\n 'favourite': 'favorite',\n 'travelling': 'traveling',\n 'counselling': 'counseling',\n 'theatre': 'theater',\n 'cancelled': 'canceled',\n 'labour': 'labor',\n 'organisation': 'organization',\n 'wwii': 'world war 2',\n 'citicise': 'criticize',\n 'youtu ': 'youtube ',\n 'Qoura': 'Quora',\n 'sallary': 'salary',\n 'Whta': 'What',\n 'narcisist': 'narcissist',\n 'howdo': 'how do',\n 'whatare': 'what are',\n 'howcan': 'how can',\n 'howmuch': 'how much',\n 'howmany': 'how many',\n 'whydo': 'why do',\n 'doI': 'do I',\n 'theBest': 'the best',\n 'howdoes': 'how does',\n 'mastrubation': 'masturbation',\n 'mastrubate': 'masturbate',\n \"mastrubating\": 'masturbating',\n 'pennis': 'penis',\n 'Etherium': 'Ethereum',\n 'narcissit': 'narcissist',\n 'bigdata': 'big data',\n '2k17': '2017',\n '2k18': '2018',\n 'qouta': 'quota',\n 'exboyfriend': 'ex boyfriend',\n 'airhostess': 'air hostess',\n \"whst\": 'what',\n 'watsapp': 'whatsapp',\n 'demonitisation': 'demonetization',\n 'demonitization': 'demonetization',\n 'demonetisation': 'demonetization'}\n\n\ndef correct_spelling(x, dic):\n    for word in dic.keys():\n        x = x.replace(word, dic[word])\n    return x\n\nprint('clean misspellings...')\ndf[\"question_text\"] = df[\"question_text\"].apply(lambda x: correct_spelling(x,mispell_dict))\ndf_final[\"question_text\"] = df_final[\"question_text\"].apply(lambda x: correct_spelling(x,mispell_dict))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1e623edf2dafa59fa7bd49cc3f7ca5587cc71918"},"cell_type":"code","source":"from keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\n\n#dim of vectors\ndim = 300\n# max words in vocab\nnum_words = 95000\n# max number in questions\nmax_len = 100 \n\nprint('Fiting tokenizer')\n## Tokenize the sentences\ntokenizer = Tokenizer(num_words=num_words)\ntokenizer.fit_on_texts(list(df['question_text'])+list(df_final['question_text']))\n\nprint('text to sequence')\nx_train = tokenizer.texts_to_sequences(df['question_text'])\n\nprint('pad sequence')\n## Pad the sentences \nx_train = pad_sequences(x_train,maxlen=max_len)\n\n## Get the target values\ny_train = df['target'].values\n\nprint(x_train.shape)\nprint(y_train.shape)\n\nx_test=tokenizer.texts_to_sequences(df_final['question_text'])\nx_test = pad_sequences(x_test,maxlen=max_len)\n\nprint('Test data loaded:',x_test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f6fa9d85ca789e88e1f674e6dae245332ec560b6"},"cell_type":"code","source":"print('Glove ... ')\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open('../input/embeddings/glove.840B.300d/glove.840B.300d.txt'))\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nprint(len(all_embs))\n\n\nword_index = tokenizer.word_index\n# num_words = min(num_words, len(word_index))\nembedding_matrix_glov = np.random.normal(emb_mean, emb_std, (num_words, dim))\ncount=0\nfor word, i in word_index.items():\n    if i >= num_words: \n        break\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: \n        embedding_matrix_glov[i] = embedding_vector\n    else:\n        count += 1\nprint('embedding matrix size:',embedding_matrix_glov.shape)\nprint('Number of words not in vocab:',count)\n\ndel embeddings_index,all_embs\nimport gc\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cba8278c1259fcb7bb918495ffef68f7dbecd95d"},"cell_type":"code","source":"print('Para...')\nEMBEDDING_FILE = '../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE, encoding=\"utf8\", errors='ignore') if len(o)>100)\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nprint(len(all_embs))\n\n\nword_index = tokenizer.word_index\n# num_words = min(num_words, len(word_index))\nembedding_matrix_para = np.random.normal(emb_mean, emb_std, (num_words, dim))\ncount=0\nfor word, i in word_index.items():\n    if i >= num_words: \n        break\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: \n        embedding_matrix_para[i] = embedding_vector\n    else:\n        count += 1\nprint('embedding matrix size:',embedding_matrix_para.shape)\nprint('Number of words not in vocab:',count)\n\ndel embeddings_index,all_embs\nimport gc\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"460f9178a497fdcf5e75ad53df0a67e3accc7fee"},"cell_type":"code","source":"#src: https://github.com/Bjarten/early-stopping-pytorch/blob/master/pytorchtools.py\nclass EarlyStopping:\n    \"\"\"Early stops the training if validation loss dosen't improve after a given patience.\"\"\"\n    def __init__(self, patience=7, verbose=False):\n        \"\"\"\n        Args:\n            patience (int): How long to wait after last time validation loss improved.\n                            Default: 7\n            verbose (bool): If True, prints a message for each validation loss improvement. \n                            Default: False\n        \"\"\"\n        self.patience = patience\n        self.verbose = verbose\n        self.counter = 0\n        self.best_score = None\n        self.early_stop = False\n        self.val_loss_min = np.Inf\n\n    def __call__(self, val_loss, model):\n\n        score = -val_loss\n\n        if self.best_score is None:\n            self.best_score = score\n            self.save_checkpoint(val_loss, model)\n        elif score < self.best_score:\n            self.counter += 1\n            if self.verbose:\n                print(f'EarlyStopping counter: {self.counter} out of {self.patience}')\n            if self.counter >= self.patience:\n                self.early_stop = True\n        else:\n            self.best_score = score\n            self.save_checkpoint(val_loss, model)\n            self.counter = 0\n\n    def save_checkpoint(self, val_loss, model):\n        '''Saves model when validation loss decrease.'''\n        if self.verbose:\n            print(f'Validation loss decreased ({self.val_loss_min:.6f} --> {val_loss:.6f}).  Saving model ...')\n        torch.save(model.state_dict(), 'checkpoint.pt')\n        self.val_loss_min = val_loss","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"90c151f3f6e8d359ff3fc710b2f25e9b66309559","scrolled":false},"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nimport torch.utils.data\nimport torchtext.data\nimport warnings\nfrom sklearn.metrics import accuracy_score\nfrom torch.autograd import Variable\nimport warnings\nfrom sklearn.model_selection import StratifiedKFold\n\n#src: https://github.com/anandsaha/pytorch.cyclic.learning.rate/blob/master/cls.py\nclass CyclicLR(object):\n    \"\"\"Sets the learning rate of each parameter group according to\n    cyclical learning rate policy (CLR). The policy cycles the learning\n    rate between two boundaries with a constant frequency, as detailed in\n    the paper `Cyclical Learning Rates for Training Neural Networks`_.\n    The distance between the two boundaries can be scaled on a per-iteration\n    or per-cycle basis.\n    Cyclical learning rate policy changes the learning rate after every batch.\n    `batch_step` should be called after a batch has been used for training.\n    To resume training, save `last_batch_iteration` and use it to instantiate `CycleLR`.\n    This class has three built-in policies, as put forth in the paper:\n    \"triangular\":\n        A basic triangular cycle w/ no amplitude scaling.\n    \"triangular2\":\n        A basic triangular cycle that scales initial amplitude by half each cycle.\n    \"exp_range\":\n        A cycle that scales initial amplitude by gamma**(cycle iterations) at each\n        cycle iteration.\n    This implementation was adapted from the github repo: `bckenstler/CLR`_\n    Args:\n        optimizer (Optimizer): Wrapped optimizer.\n        base_lr (float or list): Initial learning rate which is the\n            lower boundary in the cycle for eachparam groups.\n            Default: 0.001\n        max_lr (float or list): Upper boundaries in the cycle for\n            each parameter group. Functionally,\n            it defines the cycle amplitude (max_lr - base_lr).\n            The lr at any cycle is the sum of base_lr\n            and some scaling of the amplitude; therefore\n            max_lr may not actually be reached depending on\n            scaling function. Default: 0.006\n        step_size (int): Number of training iterations per\n            half cycle. Authors suggest setting step_size\n            2-8 x training iterations in epoch. Default: 2000\n        mode (str): One of {triangular, triangular2, exp_range}.\n            Values correspond to policies detailed above.\n            If scale_fn is not None, this argument is ignored.\n            Default: 'triangular'\n        gamma (float): Constant in 'exp_range' scaling function:\n            gamma**(cycle iterations)\n            Default: 1.0\n        scale_fn (function): Custom scaling policy defined by a single\n            argument lambda function, where\n            0 <= scale_fn(x) <= 1 for all x >= 0.\n            mode paramater is ignored\n            Default: None\n        scale_mode (str): {'cycle', 'iterations'}.\n            Defines whether scale_fn is evaluated on\n            cycle number or cycle iterations (training\n            iterations since start of cycle).\n            Default: 'cycle'\n        last_batch_iteration (int): The index of the last batch. Default: -1\n    Example:\n        >>> optimizer = torch.optim.SGD(model.parameters(), lr=0.1, momentum=0.9)\n        >>> scheduler = torch.optim.CyclicLR(optimizer)\n        >>> data_loader = torch.utils.data.DataLoader(...)\n        >>> for epoch in range(10):\n        >>>     for batch in data_loader:\n        >>>         scheduler.batch_step()\n        >>>         train_batch(...)\n    .. _Cyclical Learning Rates for Training Neural Networks: https://arxiv.org/abs/1506.01186\n    .. _bckenstler/CLR: https://github.com/bckenstler/CLR\n    \"\"\"\n\n    def __init__(self, optimizer, base_lr=1e-3, max_lr=6e-3,\n                 step_size=2000, mode='triangular', gamma=1.,\n                 scale_fn=None, scale_mode='cycle', last_batch_iteration=-1):\n\n        if not isinstance(optimizer, torch.optim.Optimizer):\n            raise TypeError('{} is not an Optimizer'.format(\n                type(optimizer).__name__))\n        self.optimizer = optimizer\n\n        if isinstance(base_lr, list) or isinstance(base_lr, tuple):\n            if len(base_lr) != len(optimizer.param_groups):\n                raise ValueError(\"expected {} base_lr, got {}\".format(\n                    len(optimizer.param_groups), len(base_lr)))\n            self.base_lrs = list(base_lr)\n        else:\n            self.base_lrs = [base_lr] * len(optimizer.param_groups)\n\n        if isinstance(max_lr, list) or isinstance(max_lr, tuple):\n            if len(max_lr) != len(optimizer.param_groups):\n                raise ValueError(\"expected {} max_lr, got {}\".format(\n                    len(optimizer.param_groups), len(max_lr)))\n            self.max_lrs = list(max_lr)\n        else:\n            self.max_lrs = [max_lr] * len(optimizer.param_groups)\n\n        self.step_size = step_size\n\n        if mode not in ['triangular', 'triangular2', 'exp_range'] \\\n                and scale_fn is None:\n            raise ValueError('mode is invalid and scale_fn is None')\n\n        self.mode = mode\n        self.gamma = gamma\n\n        if scale_fn is None:\n            if self.mode == 'triangular':\n                self.scale_fn = self._triangular_scale_fn\n                self.scale_mode = 'cycle'\n            elif self.mode == 'triangular2':\n                self.scale_fn = self._triangular2_scale_fn\n                self.scale_mode = 'cycle'\n            elif self.mode == 'exp_range':\n                self.scale_fn = self._exp_range_scale_fn\n                self.scale_mode = 'iterations'\n        else:\n            self.scale_fn = scale_fn\n            self.scale_mode = scale_mode\n\n        self.batch_step(last_batch_iteration + 1)\n        self.last_batch_iteration = last_batch_iteration\n\n    def batch_step(self, batch_iteration=None):\n        if batch_iteration is None:\n            batch_iteration = self.last_batch_iteration + 1\n        self.last_batch_iteration = batch_iteration\n        for param_group, lr in zip(self.optimizer.param_groups, self.get_lr()):\n            param_group['lr'] = lr\n\n    def _triangular_scale_fn(self, x):\n        return 1.\n\n    def _triangular2_scale_fn(self, x):\n        return 1 / (2. ** (x - 1))\n\n    def _exp_range_scale_fn(self, x):\n        return self.gamma**(x)\n\n    def get_lr(self):\n        step_size = float(self.step_size)\n        cycle = np.floor(1 + self.last_batch_iteration / (2 * step_size))\n        x = np.abs(self.last_batch_iteration / step_size - 2 * cycle + 1)\n\n        lrs = []\n        param_lrs = zip(self.optimizer.param_groups, self.base_lrs, self.max_lrs)\n        for param_group, base_lr, max_lr in param_lrs:\n            base_height = (max_lr - base_lr) * np.maximum(0, (1 - x))\n            if self.scale_mode == 'cycle':\n                lr = base_lr + base_height * self.scale_fn(cycle)\n            else:\n                lr = base_lr + base_height * self.scale_fn(self.last_batch_iteration)\n            lrs.append(lr)\n        return lrs\n\n#logistic difference equation\ndef chaos_lr(r):\n    lr_lambda = lambda iters: rel_val(iters)\n\n    def rel_val(iteration):\n        x = 0.99\n        for i in range(iteration):\n            x = x*r*(1.-x)\n#         print(x)\n        return x\n    \n    return lr_lambda\n\n#https://www.kaggle.com/shujian/single-rnn-with-4-folds-v1-9\ndef threshold_search(y_true, y_proba):\n    best_threshold = 0\n    best_score = 0\n    for threshold in [i * 0.01 for i in range(100)]:\n        with warnings.catch_warnings():\n            warnings.simplefilter(\"ignore\")\n            score = f1_score(y_true=y_true, y_pred=y_proba > threshold)\n#         print('\\rthreshold = %f | score = %f'%(threshold,score),end='')\n        if score > best_score:\n            best_threshold = threshold\n            best_score = score\n#     print('\\nbest threshold is % f with score %f'%(best_threshold,best_score))\n    search_result = {'threshold': best_threshold, 'f1': best_score}\n    return search_result\n\ntorch.cuda.init()\ntorch.cuda.empty_cache()\nprint('CUDA MEM:',torch.cuda.memory_allocated())\n\nprint('cuda:', torch.cuda.is_available())\nprint('cude index:',torch.cuda.current_device())\n\n\nbatch_size = 512\nprint('batch_size:',batch_size)\nprint('---')\n\nX_train1,X_val,y_train1,y_val = train_test_split(x_train, y_train)\nX_train2,X_test,y_train2,y_test = train_test_split(X_train1, y_train1)\n\nx_train_tensor = torch.tensor(X_train2, dtype=torch.long).cuda()\ny_train_tensor = torch.tensor(y_train2, dtype=torch.float32).cuda()\ntrain_dataset = torch.utils.data.TensorDataset(x_train_tensor,y_train_tensor)\ntrain_loader = torch.utils.data.DataLoader(dataset=train_dataset,batch_size=batch_size,shuffle=True)\n\nx_val_tensor = torch.tensor(X_val, dtype=torch.long).cuda()\ny_val_tensor = torch.tensor(y_val, dtype=torch.float32).cuda()\nval_dataset = torch.utils.data.TensorDataset(x_val_tensor,y_val_tensor)\nval_loader = torch.utils.data.DataLoader(dataset=val_dataset,batch_size=batch_size,shuffle=False)\n\n# https://www.kaggle.com/spirosrap/bilstm-attention-kfold-clr-extra-features-capsule?scriptVersionId=8688933\n# https://arxiv.org/pdf/1601.06733.pdf\nclass Attention(nn.Module):\n    def __init__(self,hidden_lstm_size):\n        super(Attention,self).__init__()\n        self.hidden_lstm_size = hidden_lstm_size\n        self.seq_len = max_len\n        weights = torch.zeros(self.hidden_lstm_size,1)\n        nn.init.xavier_uniform_(weights)\n        self.weights = nn.Parameter(weights)\n    \n    def forward(self,x):\n        #torch.Size([512, 100, 256]) -> torch.Size([51200, 256])\n        # this just makes a copy of the original 'tensor' and changes its dimension for matrix multiplication\n        x_c = x.contiguous().view(-1, self.hidden_lstm_size)\n        #If A is an n × m matrix and B is an m × p matrix, -> \n        #the matrix product C = AB (denoted without multiplication signs or dots) is defined to be the n × p matrix\n        #weights = [256,1]\n        #torch.Size([51200, 256]) x [256,1] -> torch.Size([51200, 1])\n        # REMARK: need to try using bmm instead:\n        # If batch1 is a (b×n×m) tensor, batch2 is a (b×m×p) tensor, out will be a (b×n×p) tensor.\n        a = torch.mm(x_c, self.weights)\n        # torch.Size([51200, 1]) -> torch.Size([512, 100]) \n        # thus for each word in sequence we got a 'weight'\n        a = a.view(-1, self.seq_len)\n        # torch.Size([512, 100])\n        \n        #ai^t = va^t * tanh(W*h)\n        # not using any va^t, hmm\n        # and computing the attention\n        a = torch.tanh(a)\n        #torch.Size([512, 100])        \n        # torch.Size([512, 100]) - softmax over attention of each word in seq\n        s = F.softmax(a,dim=1)\n        #torch.Size([512, 100])        \n        # [h^t,c^t] = sum(si^t*[h^i,c^i])\n        # h - hidden lstm state\n        # c - lstm memory state\n        s = torch.unsqueeze(s, -1)\n        #torch.Size([512, 100,1]) x torch.Size([512, 100, 256]) ->torch.Size([512, 100, 256])\n        weighted_input =s*x        \n        # torch.Size([512, 100, 256])-> torch.Size([512, 256])\n        return torch.sum(weighted_input, 1)\n        \n\nclass Sentiment(nn.Module):\n    \n    def __init__(self,matrix_glove,matrix_para,batch_size):\n        super(Sentiment,self).__init__()\n        print('Vocab vectors size:',matrix_glove.shape)\n        self.batch_size = batch_size\n        self.hidden_dim = 128\n        self.lin_dim  = 64\n        self.n_layers = 1 \n        \n        self.embedding_glove = nn.Embedding(matrix_glove.shape[0],matrix_glove.shape[1])\n        self.embedding_glove.weight = nn.Parameter(torch.tensor(matrix_glove, dtype=torch.float32))\n        self.embedding_glove.weight.requires_grad = False\n\n        self.embedding_para = nn.Embedding(matrix_para.shape[0],matrix_para.shape[1])\n        self.embedding_para.weight = nn.Parameter(torch.tensor(matrix_para, dtype=torch.float32))\n        self.embedding_para.weight.requires_grad = False\n        \n        self.lstm = nn.LSTM(input_size=matrix_glove.shape[1]*2, \n                            hidden_size=self.hidden_dim, \n                            num_layers=self.n_layers,\n                            bidirectional=True,\n                            batch_first=True,\n                            dropout=0.1)      \n        \n#         self.gru = nn.GRU(input_size=matrix_glove.shape[1]*2, \n#                             hidden_size=self.hidden_dim, \n#                             num_layers=self.n_layers,\n#                             bidirectional=True,\n#                             batch_first=True)      \n        \n        self.att_lstm = Attention(self.hidden_dim*2)\n#         self.att_gru = Attention(self.hidden_dim*2)\n        self.linear1 = nn.Linear(2*self.hidden_dim,self.lin_dim) \n        self.relu = nn.ReLU()\n        self.linear2 = nn.Linear(self.lin_dim,1)\n        self.dropout = nn.Dropout(0.1)\n\n        \n    def forward(self,x):\n        hidden = (torch.zeros(2*self.n_layers, x.shape[0], self.hidden_dim).cuda(),\n                torch.zeros(2*self.n_layers, x.shape[0], self.hidden_dim).cuda())\n        hidden_gru = torch.zeros(2*self.n_layers, x.shape[0], self.hidden_dim).cuda()\n        \n        e_glove = self.embedding_glove(x)\n        e_para = self.embedding_para(x)\n        e = torch.cat([e_glove,e_para],dim=-1)\n        \n        lstm_out, hidden = self.lstm(e, hidden)\n#         gru_out, hidden_gru = self.gru(e, hidden_gru)\n        out_lstm = self.att_lstm(lstm_out)\n#         out_gru = self.att_gru(gru_out)\n#         out = torch.cat([out_gru,out_lstm],dim=1)\n        \n        out = self.linear1(out_lstm)\n        out = self.relu(out)\n        out = self.dropout(out)\n        return self.linear2(out)\n        \nmodel = Sentiment(embedding_matrix_glov,embedding_matrix_para,batch_size=batch_size).cuda()\nprint(model)\nprint('-'*80)\n\nearly_stopping = EarlyStopping(patience=2,verbose=True)\nloss_function = nn.BCEWithLogitsLoss().cuda()        \noptimizer = optim.RMSprop(model.parameters(),lr=1e-3)\n# scheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer,patience=0,verbose=True)\n# scheduler = optim.lr_scheduler.LambdaLR(optimizer, [chaos_lr(3.57)])\nscheduler = CyclicLR(optimizer,base_lr=1e-3, max_lr=4e-3,\n               step_size=300., mode='exp_range',\n               gamma=0.99994)\n\n    \nlosses = []\nval_losses=[]\nepoch_acc=[]\nepoch_val_acc=[]\n\nval_list = list(val_loader)\n\nfor epoch in range(100):\n#     print('-----%d-----'%epoch)\n    epoch_losses=[]\n    epoch_val_losses = []\n    preds = []\n    val_preds=[]\n    targets = []\n    acc = []\n    for batch,(x_batch,y_true) in enumerate(list(iter(train_loader)),1):\n        model.train()\n        optimizer.zero_grad()\n        \n        y_pred = model(x_batch).squeeze(1)\n        y_numpy_pred =torch.sigmoid(y_pred).cpu().detach().numpy()\n        preds += y_numpy_pred.tolist()\n        \n        y_numpy_true = y_true.cpu().detach().numpy()\n        targets += y_numpy_true.tolist()\n        loss = loss_function(y_pred,y_true)\n        epoch_losses.append(loss.item())\n\n        loss.backward()\n        optimizer.step()\n        scheduler.batch_step()\n        acc.append(accuracy_score(y_numpy_true,np.round(y_numpy_pred)))\n        if batch % 100 == 0:\n            print('\\rtraining (batch,loss,acc) | ',batch,' ===>',loss.item(),' acc ',np.mean(acc),end='')\n    \n        \n    losses.append(np.mean(epoch_losses))\n    targets =  np.array(targets)\n    preds = np.array(preds)\n    search_result = threshold_search(targets, preds)\n    train_f1 = search_result['f1']\n    epoch_acc.append(np.mean(acc))\n    \n    targets = []\n    val_acc=[]\n    model.eval()\n    with torch.no_grad():\n        for batch,(x_val_batch,y_true) in enumerate(val_list,1):\n            y_pred = model(x_val_batch).squeeze(1)\n            y_numpy_pred = torch.sigmoid(y_pred).cpu().detach().numpy()\n            val_preds += y_numpy_pred.tolist()        \n            \n            y_numpy_true = y_true.cpu().detach().numpy()\n            targets += y_numpy_true.tolist()\n            val_loss = loss_function(y_pred,y_true)\n            epoch_val_losses.append(val_loss.item())\n            val_acc.append(accuracy_score(y_numpy_true,np.round(y_numpy_pred)))\n            if batch % 100 == 0:\n                print('\\rvalidation (batch,acc) | ',batch,' ===>', np.mean(val_acc),end='')\n    \n    val_losses.append(np.mean(epoch_val_losses))\n    epoch_val_acc.append(np.mean(val_acc))\n    \n    targets =  np.array(targets)\n    val_preds =  np.array(val_preds)\n    search_result = threshold_search(targets, val_preds)\n    val_f1 = search_result['f1']\n    \n#     scheduler.step(1.-val_f1)\n    \n    print('\\nEPOCH: ',epoch,'\\n has acc of ',epoch_acc[-1],' ,has loss of ',losses[-1], ' ,f1 of ',train_f1,'\\nval acc of ',epoch_val_acc[-1],' ,val loss of ',val_losses[-1],' ,val f1 of ',val_f1)\n    print('-'*80)\n            \n    if early_stopping.early_stop:        \n        print(\"Early stopping at \",epoch,\" epoch\")\n        break\n    else:        \n        early_stopping(1.-val_f1, model)\n\n    \nprint('Training finished....')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ae403d8c4ccc8be12768d260a65d00a912c8b119"},"cell_type":"code","source":"print(os.listdir())\n\nmodel = Sentiment(embedding_matrix_glov,embedding_matrix_para,batch_size=batch_size).cuda()\nmodel.load_state_dict(torch.load('checkpoint.pt'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d9213db94c556e3c42cdbeb741efef5c1423ae1a"},"cell_type":"code","source":"_,ax = plt.subplots(2,1,figsize=(20,10))\nax[0].plot(losses,label='loss')\nax[0].plot(val_losses,label='val_loss')\n\nax[1].plot(epoch_acc,label='acc')\nax[1].plot(epoch_val_acc,label='val_acc')\n\nplt.legend()\nplt.show()\n\nx_test_tensor = torch.tensor(X_test, dtype=torch.long).cuda()\ny_test_tensor = torch.tensor(y_test, dtype=torch.float32).cuda()\ntest_dataset = torch.utils.data.TensorDataset(x_test_tensor,y_test_tensor)\ntrain_loader = torch.utils.data.DataLoader(dataset=test_dataset,batch_size=batch_size,shuffle=True)\n\n\npred = []\ntargets = []\nwith torch.no_grad():\n    for batch,(x,y_true) in enumerate(list(train_loader),1):\n        model.eval()\n        pred += torch.sigmoid(model(x).squeeze(1)).cpu().data.numpy().tolist()\n        targets += y_true.cpu().data.numpy().tolist()\n\npred = np.array(pred)\ntargets =  np.array(targets)\nsearch_result = threshold_search(targets, pred)\npred = (pred > search_result['threshold']).astype(int)\nprint('test acc:',accuracy_score(pred,targets))\nprint('test f1:',search_result['f1'])\n\nprint('RESULTS ON TEST SET:\\n',classification_report(targets,pred))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0ad34df4ae095e1de08c64079fa6b0ecbc944423"},"cell_type":"code","source":"print('Threshold:',search_result['threshold'])\nprint(x_test.shape)\nsubmission_dataset = torch.utils.data.TensorDataset(torch.tensor(x_test, dtype=torch.long).cuda())\nsubmission_loader = torch.utils.data.DataLoader(dataset=submission_dataset,batch_size=batch_size, shuffle=False)\n\npred = []\nwith torch.no_grad():\n    for x in list(submission_loader):\n        model.eval()\n        pred += torch.sigmoid(model(x[0]).squeeze(1)).cpu().data.numpy().tolist()\n\npred = np.array(pred)\n\ndf_subm = pd.DataFrame()\ndf_subm['qid'] = df_final.qid\ndf_subm['prediction'] = (pred > search_result['threshold']).astype(int)\nprint(df_subm.head())\ndf_subm.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}