{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"#!pip install tensorflow-gpu==1.14\n#!pip install tensorflow-addons\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport operator\n#import tensorflow.compat.v1 as tf\n#tf.disable_v2_behavior()\n #v1.enable_eager_execution()\n\n\n#from tensorflow import tensorflow_addons as tfa\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nfrom sklearn import metrics\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.model_selection import train_test_split\n\nfrom keras import backend as K, initializers, regularizers, constraints, optimizers, layers\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D, concatenate\nfrom keras.layers import Bidirectional, GlobalMaxPool1D, GlobalMaxPooling1D, GlobalAveragePooling1D\nfrom keras.layers import Input, Embedding, Dense, Conv2D, MaxPool2D, concatenate, Lambda\nfrom keras.layers import Reshape, Flatten, Concatenate, Dropout, SpatialDropout1D\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.optimizers import Adam\nfrom keras.models import Model\nfrom keras.engine.topology import Layer     \n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\nprint(os.listdir(\"../input\"))\n\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"#Procesamiento de datos\n\n#Definicion del dataset\ntrain = pd.read_csv(\"../input/quora-insincere-questions-classification/train.csv\")\ntest = pd.read_csv(\"../input/quora-insincere-questions-classification/test.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Se definen los parametros para el procesamiento de los datos\n# Este es el tamaño de vector de palabras\ntam_vector = 300 \n# Cantidad máxima de palabras a tomar en cuenta. Si el vocabulario supera este número, Keras escoge las palabras con mayor frecuencia. \nmax_palabras = 100000 \n # Este sera el tamaño maximo de la pregunta\nmax_pregunta = 40\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Se construye el diccionario de palabras creando un diccionario, donde la llave es la palabra, y el valor es la frecuencia de esa palabra. \ndef crear_vocab(texts):  \n    oraciones = texts.apply(lambda x: x.split()).values #Dividimos las oraciones en una lista de arreglos, donde cada arreglo representa una oración. Las celdas en ese arreglo son palabras.\n    vocab = {}\n    for oracion in oraciones:\n        for palabra in oracion:\n            #Contamos cada palabra en por oración. Si existe en el vocabulario, se le suma 1 a las veces que se repite. Si no existe, se agrega al vocabulario con 1. \n            try:\n                vocab[palabra] += 1\n            except KeyError:\n                vocab[palabra] = 1\n    return vocab\n\n\ndf = pd.concat([train ,test], sort=False)\n\nvocab = crear_vocab(df['question_text'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Se crea una funcion para cargar la matriz de embeddings y sus indices \n\ndef cargar_embedding(file):\n    def get_coeficientes(palabra,*arr): \n        return palabra, np.asarray(arr, dtype='float32')\n    embeddings_index = dict(get_coeficientes(*o.split(\" \")) for o in open(file, encoding='latin'))\n    return embeddings_index\n\nglove = '../input/quora-insincere-questions-classification/embeddings/glove.840B.300d/glove.840B.300d.txt'\n\nembed_glove = cargar_embedding(glove)\n#se imprime la longitud de glove\nlen(embed_glove)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Se cra una funcion para cargar la matriz de glove\n\ndef cargar_matriz_glove(word_index, embeddings_index):\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean, emb_std = all_embs.mean(), all_embs.std()\n    tam_vector = all_embs.shape[1]\n    \n    nb_words = min(max_palabras, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, tam_vector))\n\n    for palabra, i in word_index.items():\n        if i >= max_palabras:\n            continue\n        embedding_vector = embeddings_index.get(palabra)\n        if embedding_vector is not None:\n            embedding_matrix[i] = embedding_vector\n\n    return embedding_matrix","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Funcion para evaluar el coverage entre el diccionacio y un conjunto de embedding\n\n\n#La funcion calcula la intersección entre el diccionario y los embeddings\n#Si hay una palabra del diccionario que no este en el embedding se devuelve como desconocida. \n\ndef check_coverage(vocab, embeddings_index):\n    pal_conocida = {}\n    pal_desconocida = {}\n    nb_pal_conocida = 0\n    nb_pal_desconocida = 0\n    for palabra in vocab.keys():\n        try:\n            pal_conocida[palabra] = embeddings_index[palabra]\n            nb_pal_conocida += vocab[palabra]\n        except:\n            pal_desconocida[palabra] = vocab[palabra]\n            nb_pal_desconocida += vocab[palabra]\n            pass\n    \n    \n    print('Se encontraron embeddings para el {:.3%} del diccionario'.format(len(pal_conocida)/len(vocab)))\n    print('Se encontraron embeddings para el {:.3%} de todo el cuerpo de texto'.format(nb_pal_conocida/(nb_pal_conocida + nb_pal_desconocida)))\n    pal_desconocida = sorted(pal_desconocida.items(), key=operator.itemgetter(1))[::-1]\n\n    return pal_desconocida","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pal_desconocida = check_coverage(vocab, embed_glove)\n\n#Para mejorar el modelo es util observar que palabras no estan en el diccionario\npal_desconocida[:20]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Funcion para cambiar a minusculas el embedding\ndef agregar_minusculas(embedding, vocab):\n    count = 0\n    for palabra in vocab:\n        if palabra in embedding and palabra.lower() not in embedding:  \n            embedding[palabra.lower()] = embedding[palabra]\n            count += 1\n    print(f\"Se agregaron {count} palabras al embedding\")\n    \n#se cambia todo a minusculas\ntrain['question_text'] = train['question_text'].apply(lambda x: x.lower())\ntest['question_text'] = test['question_text'].apply(lambda x: x.lower())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Previo\npal_desconocida = check_coverage(vocab, embed_glove)\n#Actualizado\nagregar_minusculas(embed_glove, vocab) \npal_desconocida = check_coverage(vocab, embed_glove)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Definiendo el diccionario para mapear las contracciones segun el enlace anterior\ncontraction_mapping = {\"ain't\": \"is not\", \"aren't\": \"are not\",\"can't\": \"cannot\", \"'cause\": \"because\", \"could've\": \"could have\",\n                       \"couldn't\": \"could not\", \"didn't\": \"did not\",  \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\",\n                       \"hasn't\": \"has not\", \"haven't\": \"have not\", \"he'd\": \"he would\",\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\",\n                       \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\",  \"I'd\": \"I would\", \"I'd've\": \"I would have\",\n                       \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"I've\": \"I have\", \"i'd\": \"i would\", \"i'd've\": \"i would have\",\n                       \"i'll\": \"i will\",  \"i'll've\": \"i will have\",\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \n                       \"it'd've\": \"it would have\", \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\", \n                       \"ma'am\": \"madam\", \"mayn't\": \"may not\", \"might've\": \"might have\",\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \n                       \"must've\": \"must have\", \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \"needn't\": \"need not\", \n                       \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\", \"oughtn't've\": \"ought not have\", \n                       \"shan't\": \"shall not\", \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\", \n                       \"she'll\": \"she will\", \"she'll've\": \"she will have\", \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\",\n                       \"shouldn't've\": \"should not have\", \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \n                       \"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \"there'd've\": \"there would have\", \n                       \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\", \"they'll\": \"they will\",\n                       \"they'll've\": \"they will have\", \"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\", \n                       \"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\", \"we'll've\": \"we will have\", \"we're\": \"we are\",\n                       \"we've\": \"we have\", \"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\", \"what're\": \"what are\", \n                       \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \"when've\": \"when have\", \"where'd\": \"where did\", \n                       \"where's\": \"where is\", \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \n                       \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \n                       \"won't've\": \"will not have\", \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \n                       \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\n                       \"y'all've\": \"you all have\",\"you'd\": \"you would\", \"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\",\n                       \"you're\": \"you are\", \"you've\": \"you have\", 'colour': 'color', 'centre': 'center', 'favourite': 'favorite', \n                       'travelling': 'traveling', 'counselling': 'counseling', 'theatre': 'theater', 'cancelled': 'canceled', 'labour': 'labor', \n                       'organisation': 'organization', 'wwii': 'world war 2', 'citicise': 'criticize', 'youtu ': 'youtube ', 'Qoura': 'Quora', \n                       'sallary': 'salary', 'Whta': 'What', 'narcisist': 'narcissist', 'howdo': 'how do', 'whatare': 'what are', 'howcan': 'how can',\n                       'howmuch': 'how much', 'howmany': 'how many', 'whydo': 'why do', 'doI': 'do I', 'theBest': 'the best', 'howdoes': 'how does', \n                       'mastrubation': 'masturbation', 'mastrubate': 'masturbate', \"mastrubating\": 'masturbating', 'pennis': 'penis',\n                       'Etherium': 'Ethereum', 'narcissit': 'narcissist', 'bigdata': 'big data', '2k17': '2017', '2k18': '2018', 'qouta': 'quota', \n                       'exboyfriend': 'ex boyfriend', 'airhostess': 'air hostess', \"whst\": 'what', 'watsapp': 'whatsapp', 'demonitisation': 'demonetization',\n                       'demonitization': 'demonetization', 'demonetisation': 'demonetization'}\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Se crea una funcion para mapear las contracciones en ingles\ndef quitar_contracciones(texto, mapping):\n    specials = [\"’\", \"‘\", \"´\", \"`\"]\n    for s in specials:\n        texto = texto.replace(s, \"'\")\n    texto = ' '.join([mapping[t] if t in mapping else t for t in texto.split(\" \")])\n    return texto\n\n#Eliminando las contracciones\ntrain['question_text'] = train['question_text'].apply(lambda x: quitar_contracciones(x, contraction_mapping))\ntest['question_text'] = test['question_text'].apply(lambda x: quitar_contracciones(x, contraction_mapping))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#se definen los caracteres especiales...\npunct_mapping = \"/-'?!.,#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~\" + '\"\"“”’' + '∞θ÷α•à−β∅³π‘₹´°£€\\×™√²—–&'\npunct_mapping += '©^®` <→°€™› ♥←×§″′Â█½à…“★”–●â►−¢²¬░¶↑±¿▾═¦║―¥▓—‹─▒：¼⊕▼▪†■’▀¨▄♫☆é¯♦¤▲è¸¾Ã⋅‘∞∙）↓、│（»，♪╩╚³・╦╣╔╗▬❤ïØ¹≤‡√'\n\n#Funcion para obtener todos los caracteres desconocidos entre el embedding y la lista de caracteres\ndef caracteres_desconocidos(embed, punct):\n    desconocido = ''\n    for p in punct:\n        if p not in embed:\n            desconocido += p\n            desconocido += ' '\n    return desconocido\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Definiendo el diccionario para mapear los caracteres especiales\npuncts = {\"‘\": \"'\", \"´\": \"'\", \"°\": \"\", \"€\": \"e\", \"—\": \"-\", \"–\": \"-\", \"’\": \"'\", \"_\": \"-\", \"`\": \"'\", '“': '\"', '”': '\"', '“': '\"', \"£\": \"e\", '∞': 'infinity', 'θ': 'theta', '÷': '/', 'α': 'alpha', '•': '.', 'à': 'a', '−': '-', 'β': 'beta', '∅': '', '³': '3', 'π': 'pi', '…': ' '}\n\n#Funcion para eliminar caracteres desconocidos y reemplazarlos por el correspondiente\ndef eliminar_caracteres(texto, punct, mapping):\n    for p in mapping:\n        texto = texto.replace(p, mapping[p])\n    \n    for p in punct:\n        texto = texto.replace(p, f' {p} ')\n    \n    return texto\n\n#Eliminando caracteres especiales\ntrain['question_text'] = train['question_text'].apply(lambda x: eliminar_caracteres(x, punct_mapping, puncts))\ntest['question_text'] = test['question_text'].apply(lambda x: eliminar_caracteres(x, punct_mapping, puncts))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Reconstruyendo el diccionario de palabras luego de los cambios\ndf = pd.concat([train ,test], sort=False)\nvocab = crear_vocab(df['question_text'])\n\n#Imprimiendo las primeras 10 palabras desconocidas del glove\npal_desconocida = check_coverage(vocab, embed_glove)\npal_desconocida[:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Reservando un 10% para el conjunto de validacion\ntrain, val = train_test_split(train, test_size=0.2, random_state=42)\n\n#Filtrando los datos para evitar errores\nxtrain = train['question_text'].fillna('_na_').values\nxval = val['question_text'].fillna('_na_').values\nxtest = test['question_text'].fillna('_na_').values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Tokenizaremos oraciones segun el parametro max_palabras que se definio antes, es decir, 10000\ntokenizer = Tokenizer(num_words=max_palabras)\ntokenizer.fit_on_texts(list(xtrain))\n\n#Tokenizamos el conjunto de entrenamineto, validacion y pruebas\nxtrain = tokenizer.texts_to_sequences(xtrain)\nxval = tokenizer.texts_to_sequences(xval)\nxtest = tokenizer.texts_to_sequences(xtest)\nprint(xtrain[0])\n#Nos aseguraremos de que cada oracion tenga un tamaño de pregunta,el cual es 40\nxtrain = pad_sequences(xtrain, maxlen=max_pregunta)\nxval = pad_sequences(xval, maxlen=max_pregunta)\nxtest = pad_sequences(xtest, maxlen=max_pregunta)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Definiendo las salidas esperadas y mezclando el modelo para una mayor generalizacion\nytrain = train['target'].values\nyval = val['target'].values\n\n#Mezclando el conjunto de datos\nnp.random.seed(42)\n\ntrn_idx = np.random.permutation(len(xtrain))\nval_idx = np.random.permutation(len(xval))\n\nxtrain = xtrain[trn_idx]\nytrain = ytrain[trn_idx]\nxval = xval[val_idx]\nyval = yval[val_idx]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Cargando la matriz glove de embeddings\nembedding_matrix_glove = cargar_matriz_glove(tokenizer.word_index, embed_glove)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#capa de attention\nclass Attention(Layer):\n    def __init__(self, step_dim,\n                 W_regularizer=None, b_regularizer=None,\n                 W_constraint=None, b_constraint=None,\n                 bias=True, **kwargs):\n        self.supports_masking = True\n        self.init = initializers.get('glorot_uniform')\n\n        self.W_regularizer = regularizers.get(W_regularizer)\n        self.b_regularizer = regularizers.get(b_regularizer)\n\n        self.W_constraint = constraints.get(W_constraint)\n        self.b_constraint = constraints.get(b_constraint)\n\n        self.bias = bias\n        self.step_dim = step_dim\n        self.features_dim = 0\n        super(Attention, self).__init__(**kwargs)\n\n    def build(self, input_shape):\n        assert len(input_shape) == 3\n        shapeW = (input_shape[-1],)\n        shapeB = (input_shape[1],)\n        self.W = self.add_weight(shape= shapeW,\n                                 initializer=self.init,\n                                 name='{}_W'.format(self.name),\n                                 regularizer=self.W_regularizer,\n                                 constraint=self.W_constraint)\n        self.features_dim = input_shape[-1]\n\n        if self.bias:\n            self.b = self.add_weight(shape=shapeB,\n                                     initializer='zero',\n                                     name='{}_b'.format(self.name),\n                                     regularizer=self.b_regularizer,\n                                     constraint=self.b_constraint)\n        else:\n            self.b = None\n\n        self.built = True\n\n    def compute_mask(self, input, input_mask=None):\n        return None\n\n    def call(self, x, mask=None):\n        features_dim = self.features_dim\n        step_dim = self.step_dim\n\n        eij = K.reshape(K.dot(K.reshape(x, (-1, features_dim)),\n                        K.reshape(self.W, (features_dim, 1))), (-1, step_dim))\n\n        if self.bias:\n            eij += self.b\n\n        eij = K.tanh(eij)\n        a = K.exp(eij)\n\n        if mask is not None:\n            a *= K.cast(mask, K.floatx())\n\n        a /= K.cast(K.sum(a, axis=1, keepdims=True) + K.epsilon(), K.floatx())\n\n        a = K.expand_dims(a)\n        weighted_input = x * a\n        return K.sum(weighted_input, axis=1)\n\n    def compute_output_shape(self, input_shape):\n        return input_shape[0], self.features_dim\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#se crea una funcion para aplicar la metrica de f1\ndef f1(y_true, y_pred):\n\n    def recall(y_true, y_pred):\n        \n        true_positives = K.sum(K.round(K.clip(y_true*y_pred, 0, 1)))\n        possible_positives = K.sum(K.round(K.clip(y_true, 0, 1)))\n        recall = true_positives/(possible_positives + K.epsilon())\n        return recall\n\n    def precision(y_true, y_pred):\n        \n        true_positives = K.sum(K.round(K.clip(y_true*y_pred, 0, 1)))\n        predicted_positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n        precision = true_positives/(predicted_positives + K.epsilon())\n        return precision\n\n    precision = precision(y_true, y_pred)\n    recall = recall(y_true, y_pred)\n\n    return 2*((precision*recall)/(precision+recall+K.epsilon()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#se crea el modelo de lstm con attention\ndef model_lstm_att(embedding_matrix):\n    \n    inp = Input(shape=(max_pregunta,))\n    x = Embedding(max_palabras, tam_vector, weights=[embedding_matrix], trainable=False)(inp)\n    x = Bidirectional(LSTM(64, return_sequences=True))(x)\n    x = Bidirectional(LSTM(32, return_sequences=True))(x)\n    \n    att = Attention(max_pregunta)(x)\n    \n    y = Dense(32, activation='relu')(att)\n    y = Dropout(0.1)(y)\n    outp = Dense(1, activation='sigmoid')(y)    \n\n    model = Model(inputs=inp, outputs=outp)\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=[f1, \n                                                                        \"acc\"])\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Entrenamiento del modelo!\ndef train_pred(model, epochs=2):\n    \n    for e in range(epochs):\n        model.fit(xtrain, ytrain, batch_size=512, epochs=3, validation_data=(xval, yval))\n        pred_val_y = model.predict([xval], batch_size=1024, verbose=0)\n        best_thresh = 0.5\n        best_score = 0.0\n        for thresh in np.arange(0.1, 0.501, 0.01):\n            thresh = np.round(thresh, 2)\n            score = metrics.f1_score(yval, (pred_val_y > thresh).astype(int))\n            if score > best_score:\n                best_thresh = thresh\n                best_score = score\n\n        print(\"Val F1 Score: {:.4f}\".format(best_score))\n\n    pred_test_y = model.predict([xtest], batch_size=1024, verbose=0)\n\n    return pred_val_y, pred_test_y, best_score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"paragram = '../input/quora-insincere-questions-classification/embeddings/paragram_300_sl999/paragram_300_sl999.txt'\nembedding_matrix_para = cargar_matriz_glove(tokenizer.word_index, cargar_embedding(paragram))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"embedding_matrix = np.mean([embedding_matrix_glove, embedding_matrix_para], axis=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#creacion y entrenamiento del modelo\nmodel_lstm = model_lstm_att(embedding_matrix)\nmodel_lstm.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"outputs = []\npred_val_y, pred_test_y, best_score = train_pred(model_lstm, epochs=3)\noutputs.append([pred_val_y, pred_test_y, best_score, 'model_lstm_att only Glove'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#find best threshold\noutputs.sort(key=lambda x: x[2]) \nweights = [i for i in range(1, len(outputs) + 1)]\nweights = [float(i) / sum(weights) for i in weights] \n\npred_val_y = np.mean([outputs[i][0] for i in range(len(outputs))], axis = 0)\n\nthresholds = []\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    res = metrics.f1_score(yval, (pred_val_y > thresh).astype(int))\n    thresholds.append([thresh, res])\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, res))\n    \nthresholds.sort(key=lambda x: x[1], reverse=True)\nbest_thresh = thresholds[0][0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Best threshold:\", best_thresh, \"and F1 score\", thresholds[0][1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#prediciones y archivo para el submit\npred_test_y = np.mean([outputs[i][1] for i in range(len(outputs))], axis = 0)\npred_test_y = (pred_test_y > best_thresh).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub = pd.read_csv('../input/quora-insincere-questions-classification/sample_submission.csv')\nout_df = pd.DataFrame({\"qid\":sub[\"qid\"].values})\nout_df['prediction'] = pred_test_y\nout_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}