{"cells":[{"metadata":{"_uuid":"38e7d605b52588dfa82fb54def70d25e511df5bd"},"cell_type":"markdown","source":"# Inspired by:\n\n* https://www.kaggle.com/sudalairajkumar/a-look-at-different-embeddings\n* https://www.kaggle.com/shujian/single-rnn-with-4-folds-v1-9\n* http://mlexplained.com/2018/01/13/weight-normalization-and-layer-normalization-explained-normalization-in-deep-learning-part-2/\n* https://arxiv.org/abs/1607.06450\n* https://github.com/keras-team/keras/issues/3878\n* https://www.kaggle.com/lystdo/lstm-with-word2vec-embeddings\n* https://www.kaggle.com/jhoward/improved-lstm-baseline-glove-dropout\n* https://www.kaggle.com/aquatic/entity-embedding-neural-net\n* https://www.kaggle.com/hireme/fun-api-keras-f1-metric-cyclical-learning-rate\n* https://ai.google/research/pubs/pub46697\n* https://blog.openai.com/quantifying-generalization-in-reinforcement-learning/\n* https://www.kaggle.com/rasvob/let-s-try-clr-v3\n* https://github.com/bentrevett/pytorch-sentiment-analysis/blob/master/3%20-%20Faster%20Sentiment%20Analysis.ipynb\n* https://www.kaggle.com/ziliwang/pytorch-text-cnn\n* https://github.com/yunjey/pytorch-tutorial/blob/master/tutorials/02-intermediate/bidirectional_recurrent_neural_network/main.py\n* https://github.com/clairett/pytorch-sentiment-classification/blob/master/bilstm.py\n* https://pytorch.org/tutorials/beginner/nlp/sequence_models_tutorial.html\n* https://www.kaggle.com/bminixhofer/a-validation-framework-impact-of-the-random-seed?scriptVersionId=9370229\n"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport sys\nnp.set_printoptions(threshold=sys.maxsize)\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\nprint(os.listdir(\"../input/embeddings\"))\nprint(os.listdir(\"../input/embeddings/GoogleNews-vectors-negative300\"))\n\n# Any results you write to the current directory are saved as output.\nimport torch\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report,f1_score,precision_recall_fscore_support,recall_score,precision_score\nfrom sklearn.utils import class_weight\nimport matplotlib.pyplot as plt\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1cc7cdbeb6195fdb0fe469acc635c58ea0233efd"},"cell_type":"code","source":"import random\n\ndef seed_everything(seed=1234):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    \nseed_everything(6017)\nprint('Seeding done...')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"77b7c5199c59943744495e62d7c0f73f68769e17"},"cell_type":"code","source":"df = pd.read_csv('../input/train.csv')\ndf[\"question_text\"].fillna(\"_##_\",inplace=True)\nmax_len = df['question_text'].apply(lambda x:len(x)).max()\nprint('max length of sequences:',max_len)\n# df = df.sample(frac=0.1)\n\nprint('columns:',df.columns)\npd.set_option('display.max_columns',None)\nprint('df head:',df.head())\nprint('example of the question text values:',df['question_text'].head().values)\nprint('what values contains target:',df.target.unique())\n\nprint('Loading test data...')\ndf_final = pd.read_csv('../input/test.csv')\ndf_final[\"question_text\"].fillna(\"_##_\", inplace=True)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"092653c0bd0b41ac6123c452650c57e8577551cb"},"cell_type":"code","source":"import re\n\nprint('Preproccesing texts....')\n\nprint('lower...')\ndf[\"question_text\"] = df[\"question_text\"].apply(lambda x: x.lower())\ndf_final[\"question_text\"] = df_final[\"question_text\"].apply(lambda x: x.lower())\n\ncontraction_mapping = {\n\"ain't\": \"is not\", \n\"aren't\": \"are not\",\n\"can't\": \"cannot\",\n \"'cause\": \"because\",\n \"could've\": \"could have\",\n \"couldn't\": \"could not\",\n \"didn't\": \"did not\",\n  \"doesn't\": \"does not\",\n \"don't\": \"do not\",\n \"hadn't\": \"had not\",\n \"hasn't\": \"has not\",\n \"haven't\": \"have not\",\n \"he'd\": \"he would\",\n\"he'll\": \"he will\",\n \"he's\": \"he is\",\n \"how'd\": \"how did\",\n \"how'd'y\": \"how do you\",\n \"how'll\": \"how will\",\n \"how's\": \"how is\",\n  \"I'd\": \"I would\",\n \"I'd've\": \"I would have\",\n \"I'll\": \"I will\",\n \"I'll've\": \"I will have\",\n\"I'm\": \"I am\",\n \"I've\": \"I have\",\n \"i'd\": \"i would\",\n \"i'd've\": \"i would have\",\n \"i'll\": \"i will\",\n  \"i'll've\": \"i will have\",\n\"i'm\": \"i am\",\n \"i've\": \"i have\",\n \"isn't\": \"is not\",\n \"it'd\": \"it would\",\n \"it'd've\": \"it would have\",\n \"it'll\": \"it will\",\n \"it'll've\": \"it will have\",\n\"it's\": \"it is\",\n \"let's\": \"let us\",\n \"ma'am\": \"madam\",\n \"mayn't\": \"may not\",\n \"might've\": \"might have\",\n\"mightn't\": \"might not\",\n\"mightn't've\": \"might not have\",\n \"must've\": \"must have\",\n \"mustn't\": \"must not\",\n \"mustn't've\": \"must not have\",\n \"needn't\": \"need not\",\n \"needn't've\": \"need not have\",\n\"o'clock\": \"of the clock\",\n \"oughtn't\": \"ought not\",\n \"oughtn't've\": \"ought not have\",\n \"shan't\": \"shall not\",\n \"sha'n't\": \"shall not\",\n \"shan't've\": \"shall not have\",\n \"she'd\": \"she would\",\n \"she'd've\": \"she would have\",\n \"she'll\": \"she will\",\n \"she'll've\": \"she will have\",\n \"she's\": \"she is\",\n \"should've\": \"should have\",\n \"shouldn't\": \"should not\",\n \"shouldn't've\": \"should not have\",\n \"so've\": \"so have\",\n\"so's\": \"so as\",\n \"this's\": \"this is\",\n\"that'd\": \"that would\",\n \"that'd've\": \"that would have\",\n \"that's\": \"that is\",\n \"there'd\": \"there would\",\n \"there'd've\": \"there would have\",\n \"there's\": \"there is\",\n \"here's\": \"here is\",\n\"they'd\": \"they would\",\n \"they'd've\": \"they would have\",\n \"they'll\": \"they will\",\n \"they'll've\": \"they will have\",\n \"they're\": \"they are\",\n \"they've\": \"they have\",\n \"to've\": \"to have\",\n \"wasn't\": \"was not\",\n \"we'd\": \"we would\",\n \"we'd've\": \"we would have\",\n \"we'll\": \"we will\",\n \"we'll've\": \"we will have\",\n \"we're\": \"we are\",\n \"we've\": \"we have\",\n \"weren't\": \"were not\",\n \"what'll\": \"what will\",\n \"what'll've\": \"what will have\",\n \"what're\": \"what are\",\n  \"what's\": \"what is\",\n \"what've\": \"what have\",\n \"when's\": \"when is\",\n \"when've\": \"when have\",\n \"where'd\": \"where did\",\n \"where's\": \"where is\",\n \"where've\": \"where have\",\n \"who'll\": \"who will\",\n \"who'll've\": \"who will have\",\n \"who's\": \"who is\",\n \"who've\": \"who have\",\n \"why's\": \"why is\",\n \"why've\": \"why have\",\n \"will've\": \"will have\",\n \"won't\": \"will not\",\n \"won't've\": \"will not have\",\n \"would've\": \"would have\",\n \"wouldn't\": \"would not\",\n \"wouldn't've\": \"would not have\",\n \"y'all\": \"you all\",\n \"y'all'd\": \"you all would\",\n\"y'all'd've\": \"you all would have\",\n\"y'all're\": \"you all are\",\n\"y'all've\": \"you all have\",\n\"you'd\": \"you would\",\n \"you'd've\": \"you would have\",\n \"you'll\": \"you will\",\n \"you'll've\": \"you will have\",\n \"you're\": \"you are\",\n \"you've\": \"you have\" }\n\ndef clean_contractions(text, mapping):\n    specials = [\"’\", \"‘\", \"´\", \"`\"]\n    for s in specials:\n        text = text.replace(s, \"'\")\n    text = ' '.join([mapping[t] if t in mapping else t for t in text.split(\" \")])\n    return text\n\nprint('contractions...')\ndf[\"question_text\"] = df[\"question_text\"].apply(lambda x: clean_contractions(x,contraction_mapping))\ndf_final[\"question_text\"] = df_final[\"question_text\"].apply(lambda x: clean_contractions(x,contraction_mapping))\n\n\npunct_mapping = {\"‘\": \"'\",\n \"₹\": \"e\",\n \"´\": \"'\",\n \"°\": \"\",\n \"€\": \"e\",\n \"™\": \"tm\",\n \"√\": \" sqrt \",\n \"×\": \"x\",\n \"²\": \"2\",\n \"—\": \"-\",\n \"–\": \"-\",\n \"’\": \"'\",\n \"_\": \"-\",\n \"`\": \"'\",\n '“': '\"',\n '”': '\"',\n '“': '\"',\n \"£\": \"e\",\n '∞': 'infinity',\n 'θ': 'theta',\n '÷': '/',\n 'α': 'alpha',\n '•': '.',\n 'à': 'a',\n '−': '-',\n 'β': 'beta',\n '∅': '',\n '³': '3',\n 'π': 'pi',\n }\n\npunct = [',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\', '•',  '~', '@', '£', \n '·', '_', '{', '}', '©', '^', '®', '`',  '<', '→', '°', '€', '™', '›',  '♥', '←', '×', '§', '″', '′', 'Â', '█', '½', 'à', '…', \n '“', '★', '”', '–', '●', 'â', '►', '−', '¢', '²', '¬', '░', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥', '▓', '—', '‹', '─', \n '▒', '：', '¼', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', 'é', '¯', '♦', '¤', '▲', 'è', '¸', '¾', 'Ã', '⋅', '‘', '∞', \n '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '³', '・', '╦', '╣', '╔', '╗', '▬', '❤', 'ï', 'Ø', '¹', '≤', '‡', '√', ]\n\ndef clean_special_chars(text, punct, mapping):\n    for p in mapping:\n        text = text.replace(p, mapping[p])\n    \n    for p in punct:\n        text = text.replace(p, f' {p} ')\n    \n    specials = {'\\u200b': ' ', '…': ' ... ', '\\ufeff': '', 'करना': '', 'है': ''}  # Other special characters that I have to deal with in last\n    for s in specials:\n        text = text.replace(s, specials[s])\n    \n    return text\n\nprint('clean special chars...')\ndf[\"question_text\"] = df[\"question_text\"].apply(lambda x: clean_special_chars(x, punct, punct_mapping))\ndf_final[\"question_text\"] = df_final[\"question_text\"].apply(lambda x: clean_special_chars(x, punct, punct_mapping))\n\nmispell_dict = {'colour': 'color',\n 'centre': 'center',\n 'favourite': 'favorite',\n 'travelling': 'traveling',\n 'counselling': 'counseling',\n 'theatre': 'theater',\n 'cancelled': 'canceled',\n 'labour': 'labor',\n 'organisation': 'organization',\n 'wwii': 'world war 2',\n 'citicise': 'criticize',\n 'youtu ': 'youtube ',\n 'Qoura': 'Quora',\n 'sallary': 'salary',\n 'Whta': 'What',\n 'narcisist': 'narcissist',\n 'howdo': 'how do',\n 'whatare': 'what are',\n 'howcan': 'how can',\n 'howmuch': 'how much',\n 'howmany': 'how many',\n 'whydo': 'why do',\n 'doI': 'do I',\n 'theBest': 'the best',\n 'howdoes': 'how does',\n 'mastrubation': 'masturbation',\n 'mastrubate': 'masturbate',\n \"mastrubating\": 'masturbating',\n 'pennis': 'penis',\n 'Etherium': 'Ethereum',\n 'narcissit': 'narcissist',\n 'bigdata': 'big data',\n '2k17': '2017',\n '2k18': '2018',\n 'qouta': 'quota',\n 'exboyfriend': 'ex boyfriend',\n 'airhostess': 'air hostess',\n \"whst\": 'what',\n 'watsapp': 'whatsapp',\n 'demonitisation': 'demonetization',\n 'demonitization': 'demonetization',\n 'demonetisation': 'demonetization'}\n\n\ndef correct_spelling(x, dic):\n    for word in dic.keys():\n        x = x.replace(word, dic[word])\n    return x\n\nprint('clean misspellings...')\ndf[\"question_text\"] = df[\"question_text\"].apply(lambda x: correct_spelling(x,mispell_dict))\ndf_final[\"question_text\"] = df_final[\"question_text\"].apply(lambda x: correct_spelling(x,mispell_dict))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f41b23c1f3f4eed0d8d419974fe795b63f3df50b"},"cell_type":"code","source":"from keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\n\n#dim of vectors\ndim = 300\n# max words in vocab\nnum_words = 75966\n# max number in questions\nmax_len = 100 \n\nprint('Fiting tokenizer')\n## Tokenize the sentences\ntokenizer = Tokenizer(num_words=num_words)\ntokenizer.fit_on_texts(list(df['question_text'])+list(df_final['question_text']))\n\nprint('text to sequence')\nx_train = tokenizer.texts_to_sequences(df['question_text'])\n\nprint('pad sequence')\n## Pad the sentences \nx_train = pad_sequences(x_train,maxlen=max_len)\n\n## Get the target values\ny_train = df['target'].values\n\nprint(x_train.shape)\nprint(y_train.shape)\n\nx_test=tokenizer.texts_to_sequences(df_final['question_text'])\nx_test = pad_sequences(x_test,maxlen=max_len)\n\nprint('Test data loaded:',x_test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0f1ed31984c07cbb1a95c250e0dadf9eb649e5a3"},"cell_type":"code","source":"print('Glove ... ')\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open('../input/embeddings/glove.840B.300d/glove.840B.300d.txt'))\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nprint(len(all_embs))\n\n\nword_index = tokenizer.word_index\n# num_words = min(num_words, len(word_index))\nembedding_matrix_glov = np.random.normal(emb_mean, emb_std, (num_words, dim))\ncount=0\nfor word, i in word_index.items():\n    if i >= num_words: \n        break\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: \n        embedding_matrix_glov[i] = embedding_vector\n    else:\n        count += 1\nprint('embedding matrix size:',embedding_matrix_glov.shape)\nprint('Number of words not in vocab:',count)\n\ndel embeddings_index,all_embs\nimport gc\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9712758dc3cab221d6e57f43e5eb00224386d7d5"},"cell_type":"code","source":"print('Para...')\nEMBEDDING_FILE = '../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE, encoding=\"utf8\", errors='ignore') if len(o)>100)\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nprint(len(all_embs))\n\n\nword_index = tokenizer.word_index\n# num_words = min(num_words, len(word_index))\nembedding_matrix_para = np.random.normal(emb_mean, emb_std, (num_words, dim))\ncount=0\nfor word, i in word_index.items():\n    if i >= num_words: \n        break\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: \n        embedding_matrix_para[i] = embedding_vector\n    else:\n        count += 1\nprint('embedding matrix size:',embedding_matrix_para.shape)\nprint('Number of words not in vocab:',count)\n\ndel embeddings_index,all_embs\nimport gc\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d17e4323bef806ca6fa47c1edf8a3764a082caa3"},"cell_type":"code","source":"matrixes = [embedding_matrix_glov,embedding_matrix_para]\n\nmatrix = np.mean(matrixes,axis=0)\n\ndel embedding_matrix_glov,embedding_matrix_para\nimport gc\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c8a42fe0db821ca31d7b2f31bc3923144aa14b74"},"cell_type":"code","source":"#src: https://github.com/Bjarten/early-stopping-pytorch/blob/master/pytorchtools.py\nclass EarlyStopping:\n    \"\"\"Early stops the training if validation loss dosen't improve after a given patience.\"\"\"\n    def __init__(self, patience=7, verbose=False):\n        \"\"\"\n        Args:\n            patience (int): How long to wait after last time validation loss improved.\n                            Default: 7\n            verbose (bool): If True, prints a message for each validation loss improvement. \n                            Default: False\n        \"\"\"\n        self.patience = patience\n        self.verbose = verbose\n        self.counter = 0\n        self.best_score = None\n        self.early_stop = False\n        self.val_loss_min = np.Inf\n\n    def __call__(self, val_loss, model):\n\n        score = -val_loss\n\n        if self.best_score is None:\n            self.best_score = score\n            self.save_checkpoint(val_loss, model)\n        elif score < self.best_score:\n            self.counter += 1\n            if self.verbose:\n                print(f'EarlyStopping counter: {self.counter} out of {self.patience}')\n            if self.counter >= self.patience:\n                self.early_stop = True\n        else:\n            self.best_score = score\n            self.save_checkpoint(val_loss, model)\n            self.counter = 0\n\n    def save_checkpoint(self, val_loss, model):\n        '''Saves model when validation loss decrease.'''\n        if self.verbose:\n            print(f'Validation loss decreased ({self.val_loss_min:.6f} --> {val_loss:.6f}).  Saving model ...')\n        torch.save(model.state_dict(), 'checkpoint.pt')\n        self.val_loss_min = val_loss","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"55cb2ac1dca7de9fba51e8a7e5dba402159be302","scrolled":true},"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nimport torch.utils.data\nimport torchtext.data\nimport warnings\nfrom sklearn.metrics import accuracy_score\nfrom torch.autograd import Variable\nimport warnings\nfrom sklearn.model_selection import StratifiedKFold\n\n#https://www.kaggle.com/shujian/single-rnn-with-4-folds-v1-9\ndef threshold_search(y_true, y_proba):\n    best_threshold = 0\n    best_score = 0\n    for threshold in [i * 0.01 for i in range(100)]:\n        with warnings.catch_warnings():\n            warnings.simplefilter(\"ignore\")\n            score = f1_score(y_true=y_true, y_pred=y_proba > threshold)\n#         print('\\rthreshold = %f | score = %f'%(threshold,score),end='')\n        if score > best_score:\n            best_threshold = threshold\n            best_score = score\n#     print('\\nbest threshold is % f with score %f'%(best_threshold,best_score))\n    search_result = {'threshold': best_threshold, 'f1': best_score}\n    return search_result\n\nclass Sentiment(nn.Module):\n    \n    def __init__(self,matrix,batch_size):\n        super(Sentiment,self).__init__()\n        print('Vocab vectors size:',matrix.shape)\n        self.batch_size = batch_size\n        self.hidden_dim = 64\n        self.lin_dim  = 32\n        self.n_layers = 1 \n        \n        self.embedding = nn.Embedding(matrix.shape[0],matrix.shape[1])\n        self.embedding.weight = nn.Parameter(torch.tensor(matrix, dtype=torch.float32))\n        self.embedding.weight.requires_grad = False\n        \n        self.lstm = nn.LSTM(input_size=matrix.shape[1], \n                            hidden_size=self.hidden_dim, \n                            num_layers=self.n_layers,\n                            bidirectional=True,\n                            batch_first=True)      \n        \n        self.gru = nn.GRU(input_size=matrix.shape[1], \n                            hidden_size=self.hidden_dim, \n                            num_layers=self.n_layers,\n                            bidirectional=True,\n                            batch_first=True)   \n        \n        self.linear1 = nn.Linear(4*self.hidden_dim,self.lin_dim) \n        self.relu = nn.ReLU()\n        self.linear2 = nn.Linear(self.lin_dim,1)\n        self.dropout = nn.Dropout(0.1)\n\n        \n    def forward(self,x):\n        hidden = (torch.zeros(2*self.n_layers, x.shape[0], self.hidden_dim).cuda(),\n                torch.zeros(2*self.n_layers, x.shape[0], self.hidden_dim).cuda())\n        hidden_gru = torch.zeros(2*self.n_layers, x.shape[0], self.hidden_dim).cuda()\n        \n        e = self.embedding(x)\n        \n        lstm_out, hidden = self.lstm(e, hidden)\n        gru_out, hidden_gru = self.gru(e, hidden_gru)\n        out = torch.cat([hidden[0][-2,:,:], hidden[0][-1,:,:],hidden_gru[-2,:,:], hidden_gru[-1,:,:]], dim=1).cuda()        \n        \n        out = self.linear1(out)\n        out = self.relu(out)\n        out = self.dropout(out)        \n        return self.linear2(out)\n\n    \ndef train(model, train_loader ):\n    loss_function = nn.BCEWithLogitsLoss().cuda()        \n    optimizer = optim.Adam(model.parameters(),lr=1e-3)\n    \n    for epoch in range(3):\n        model.train()\n        avg_loss = 0\n        for batch,(x_batch,y_true) in enumerate(list(iter(train_loader)),1):\n            optimizer.zero_grad()\n\n            y_pred = model(x_batch).squeeze(1)\n            loss = loss_function(y_pred,y_true)\n            avg_loss += loss.item()\n            \n            loss.backward()\n            optimizer.step()\n            \n        print('EPOCH: ',epoch,': ',avg_loss/batch)\n        print('-'*80)\n\n    print('Training finished....')    \n\ndef eval_on_set(test_loader):\n    pred = []\n    with torch.no_grad():\n        model.eval()\n        for x_test_batch,y_test_batch in list(test_loader):\n            pred += torch.sigmoid(model(x_test_batch).squeeze(1)).cpu().detach().numpy().tolist()\n\n    return np.array(pred)\n    \ndef eval_sub(submission_loader):\n\n    pred = []\n    with torch.no_grad():\n        model.eval()\n        for (x,) in list(submission_loader):       \n            y_pred = torch.sigmoid(model(x).squeeze(1)).detach()\n            pred += y_pred.cpu().numpy().tolist()\n\n    return np.array(pred)\n    \n\nbatch_size = 512\nprint('Batch size = ',batch_size)\n\nsubmission_dataset = torch.utils.data.TensorDataset(torch.tensor(x_test, dtype=torch.long).cuda())\nsubmission_loader = torch.utils.data.DataLoader(dataset=submission_dataset,batch_size=batch_size, shuffle=False)\n\n\npatience = 2\n\ntrain_meta = np.zeros(y_train.shape)\ntest_meta = np.zeros(x_test.shape[0])\n\nsplits = list(StratifiedKFold(n_splits=4, shuffle=True, random_state=6017).split(x_train, y_train))\n\nfor idx, (train_idx, valid_idx) in enumerate(splits):\n    print('----'+str(idx)+'-----')\n    X_train1 = x_train[train_idx]\n    y_train1 = y_train[train_idx]\n    X_val = x_train[valid_idx]\n    y_val = y_train[valid_idx]\n    \n    x_train_tensor = torch.tensor(X_train1, dtype=torch.long).cuda()\n    y_train_tensor = torch.tensor(y_train1, dtype=torch.float32).cuda()\n    train_dataset = torch.utils.data.TensorDataset(x_train_tensor,y_train_tensor)\n    train_loader = torch.utils.data.DataLoader(dataset=train_dataset,batch_size=batch_size,shuffle=True)\n    \n    x_val_tensor = torch.tensor(X_val, dtype=torch.long).cuda()\n    y_val_tensor = torch.tensor(y_val, dtype=torch.float32).cuda()\n    val_dataset = torch.utils.data.TensorDataset(x_val_tensor,y_val_tensor)\n    val_loader = torch.utils.data.DataLoader(dataset=val_dataset,batch_size=batch_size,shuffle=False)\n    \n    \n    model = Sentiment(matrix, batch_size=batch_size).cuda()    \n    train(model,train_loader)\n        \n    y_pred = eval_on_set(val_loader)\n    train_meta[valid_idx] = y_pred.reshape(-1)\n        \n    #for test set\n    y_pred = eval_sub(submission_loader)\n    test_meta += y_pred.reshape(-1) / len(splits)\n\nprint('FINISHED TRAINING META...')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0ad34df4ae095e1de08c64079fa6b0ecbc944423"},"cell_type":"code","source":"#submission\nsearch_result = threshold_search(y_train, train_meta)\nprint(search_result)\n\ndf_subm = pd.DataFrame()\ndf_subm['qid'] = df_final.qid\ndf_subm['prediction'] = (test_meta > search_result['threshold']).astype(int)\nprint(df_subm.head())\ndf_subm.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}},"nbformat":4,"nbformat_minor":1}