{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"#Importando las librerias \n\nimport operator\n\nfrom sklearn import metrics\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.model_selection import train_test_split\n\nfrom keras import backend as K, initializers, regularizers, constraints, optimizers, layers\nfrom keras.layers import Dense, Input,LSTM, Embedding, Dropout, Activation,GRU, Conv1D, concatenate\nfrom keras.layers import Bidirectional, GlobalMaxPool1D, GlobalMaxPooling1D, GlobalAveragePooling1D\nfrom keras.layers import Input, Embedding, Dense, Conv2D, MaxPool2D, concatenate, Lambda\nfrom keras.layers import Reshape, Flatten, Concatenate, Dropout, SpatialDropout1D\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.optimizers import Adam\nfrom keras.models import Model\nfrom keras.engine.topology import Layer\n\nimport os\nprint(os.listdir(\"../input\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Ahora se hace el procesamiento de los datos\nentrenamiento = pd.read_csv(\"../input/quora-insincere-questions-classification/train.csv\")\nprueba = pd.read_csv(\"../input/quora-insincere-questions-classification/test.csv\")\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Definición de parámetros\ntamano_embedding=300\nMAX_FEATURES=100000 #esta es la cantidad máxima de palabras a tomar en cuenta\nMAXLEN =40 #la longitud maxima de la pregunta será 40","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Construccion del diccionario \n\ndef construccion_diccionario(texto): \n    #ahora se procede a separar cada oración en una lista de arreglos, para cada oración hay un arreglo y las celdas de estos las ocupan las palabras\n    oraciones=texto.apply(lambda x: x.split()).values\n    diccionario={}\n    \n    #Se procede a contar cada palabra en cada una de las oraciones\n    for oracion in oraciones:\n        for palabra in oracion:\n            \n            try: \n                diccionario[palabra]+=1 #si la palabra existe en el diccionario se le suma uno a las veces que esta se repite\n            except KeyError: \n                diccionario[palabra]=1 #pero, si la palabra no existe en el diccionario se agrega al diccionario con 1\n    return diccionario\n\ndf = pd.concat([entrenamiento ,prueba], sort=False)\n\ndiccionario = construccion_diccionario(df['question_text'])\nprint(\"Tamaño inicial del diccionario:\")\nprint(len(diccionario)) #imprimiendo el tamano inicial del diccionario\n#Imprimiendo los primeros 10 elementos del diccionario\nfor x in list(diccionario)[0:10]:\n    print (x, diccionario[x])\n    print()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#A continuacion se define la funcion cargar_embedding para cargar la matriz de embeddings y sus indices\ndef cargar_embedding(file):\n    def obtener_coeficientes(palabra,*arr): \n        return palabra, np.asarray(arr, dtype='float32')\n    indice_embeddings = dict(obtener_coeficientes(*o.split(\" \")) for o in open(file, encoding='latin'))\n    return indice_embeddings\n\nglove = '../input/quora-insincere-questions-classification/embeddings/glove.840B.300d/glove.840B.300d.txt'\n\nembed_glove = cargar_embedding(glove)\n\nprint('Glove embeddings cargados!')\nlen(embed_glove)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Se define una funcion para cargar la matriz de glove\n\ndef cargar_m_glove(indice_palabra, indice_embedding):\n\n    all_embs = np.stack(indice_embedding.values())\n    emb_mean, emb_std = all_embs.mean(), all_embs.std()\n    tamano_embedding = all_embs.shape[1]\n    \n    nb_palabras = min(MAX_FEATURES, len(indice_palabra))\n    matriz_embedding = np.random.normal(emb_mean, emb_std, (nb_palabras, tamano_embedding))\n\n    for palabra, i in indice_palabra.items():\n        if i >= MAX_FEATURES:\n            continue\n        vector_embedding = indice_embedding.get(palabra)\n        if vector_embedding is not None:\n           matriz_embedding[i] = vector_embedding\n\n    return matriz_embedding","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#se define una funcion para determinar el coverage entre el diccionario y un conjunto de embedding\n\ndef check_coverage(diccionario,indice_embedding):\n    palabras_conocidas = {}\n    palabras_desconocidas = {}\n    nb_palabras_conocidas = 0\n    nb_palabras_desconocidas = 0\n    for palabra in diccionario.keys():\n        try:\n            palabras_conocidas[palabra] = indice_embedding[palabra]\n            nb_palabras_conocidas += diccionario[palabra]\n        except:\n             palabras_desconocidas[palabra] = diccionario[palabra]\n             nb_palabras_desconocidas+= diccionario[palabra]\n             pass\n    \n    print(\"Glove\")\n    print('Se encontraron embeddings para el {:.3%} del diccionario'.format(len(palabras_conocidas)/len(diccionario)))\n    print('Se encontraron embeddings para el {:.3%} de todo el cuerpo de texto'.format(nb_palabras_conocidas/(nb_palabras_conocidas + nb_palabras_desconocidas)))\n    palabras_desconocidas = sorted(palabras_desconocidas.items(), key=operator.itemgetter(1))[::-1]\n\n    return palabras_desconocidas","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"palabras_desconocidas = check_coverage(diccionario, embed_glove)\n\n#Es util observar que palabras no estan en el diccionario, para mejorar el modelo\npalabras_desconocidas[:20]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#definimos una funcion para agregar minusculas al embedding\ndef agregar_minusculas(embedding, diccionario):\n    count = 0\n    for palabra in diccionario:\n        if palabra in embedding and palabra.lower() not in embedding:  \n            embedding[palabra.lower()] = embedding[palabra]\n            count += 1\n    print(f\"Added {count} words to embedding\")\n    \n#se lleva todo a minusculas\nentrenamiento['question_text'] = entrenamiento['question_text'].apply(lambda x: x.lower())\nprueba['question_text'] = prueba['question_text'].apply(lambda x: x.lower())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Imprimiendo el Glove!\")\n#Previo\npalabras_desconocidas = check_coverage(diccionario, embed_glove)\n\n#Actualizamos\nagregar_minusculas(embed_glove, diccionario) \npalabras_desconocidas = check_coverage(diccionario, embed_glove)\n\n#Imprimimos 10 palabras desconocidas del glove\npalabras_desconocidas[:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#se define el diccionario para mapear las contracciones de este enlace:  https://www.kaggle.com/c/quora-insincere-questions-classification/discussion/77758\nmapeo_contracciones = {\"ain't\": \"is not\", \"aren't\": \"are not\",\"can't\": \"cannot\", \"'cause\": \"because\", \"could've\": \"could have\", \"couldn't\": \"could not\", \"didn't\": \"did not\",  \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\", \"he'd\": \"he would\",\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\", \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\",  \"I'd\": \"I would\", \"I'd've\": \"I would have\", \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"I've\": \"I have\", \"i'd\": \"i would\", \"i'd've\": \"i would have\", \"i'll\": \"i will\",  \"i'll've\": \"i will have\",\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'd've\": \"it would have\", \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\", \"ma'am\": \"madam\", \"mayn't\": \"may not\", \"might've\": \"might have\",\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \"must've\": \"must have\", \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \"needn't\": \"need not\", \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\", \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\", \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\", \"she'll\": \"she will\", \"she'll've\": \"she will have\", \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\", \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \"there'd've\": \"there would have\", \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\", \"they'll\": \"they will\", \"they'll've\": \"they will have\", \"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\", \"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\", \"we'll've\": \"we will have\", \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\", \"what're\": \"what are\",  \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\", \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\"you'd\": \"you would\", \"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\", 'colour': 'color', 'centre': 'center', 'favourite': 'favorite', 'travelling': 'traveling', 'counselling': 'counseling', 'theatre': 'theater', 'cancelled': 'canceled', 'labour': 'labor', 'organisation': 'organization', 'wwii': 'world war 2', 'citicise': 'criticize', 'youtu ': 'youtube ', 'Qoura': 'Quora', 'sallary': 'salary', 'Whta': 'What', 'narcisist': 'narcissist', 'howdo': 'how do', 'whatare': 'what are', 'howcan': 'how can', 'howmuch': 'how much', 'howmany': 'how many', 'whydo': 'why do', 'doI': 'do I', 'theBest': 'the best', 'howdoes': 'how does', 'mastrubation': 'masturbation', 'mastrubate': 'masturbate', \"mastrubating\": 'masturbating', 'pennis': 'penis', 'Etherium': 'Ethereum', 'narcissit': 'narcissist', 'bigdata': 'big data', '2k17': '2017', '2k18': '2018', 'qouta': 'quota', 'exboyfriend': 'ex boyfriend', 'airhostess': 'air hostess', \"whst\": 'what', 'watsapp': 'whatsapp', 'demonitisation': 'demonetization', 'demonitization': 'demonetization', 'demonetisation': 'demonetization'}\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#definimos una funcion para mapear las contracciones en ingles\ndef quitar_contracciones(texto, mapeo):\n    especiales = [\"’\", \"‘\", \"´\", \"`\"]\n    for s in especiales:\n        texto = texto.replace(s, \"'\")\n    texto = ' '.join([mapeo[t] if t in mapeo else t for t in texto.split(\" \")])\n    return texto\n\n#Eliminando las contracciones\nentrenamiento['question_text'] = entrenamiento['question_text'].apply(lambda x: quitar_contracciones(x, mapeo_contracciones))\nprueba['question_text'] = prueba['question_text'].apply(lambda x: quitar_contracciones(x, mapeo_contracciones))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Se reconstruye el diccionario de palabras para guardar los cambios\ndf = pd.concat([entrenamiento ,prueba], sort=False)\ndiccionario = construccion_diccionario(df['question_text'])\n\n#se imprimen las primeras 10 palabras desconocidas del glove\nprint(\"Glove: \")\npalabras_desconocidas = check_coverage(diccionario, embed_glove)\npalabras_desconocidas[:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Eliminando caracteres especiales\n#se definen los caracteres especiales\nmapeo_puntuacion = \"/-'?!.,#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~\" + '\"\"“”’' + '∞θ÷α•à−β∅³π‘₹´°£€\\×™√²—–&'\nmapeo_puntuacion += '©^®` <→°€™› ♥←×§″′Â█½à…“★”–●â►−¢²¬░¶↑±¿▾═¦║―¥▓—‹─▒：¼⊕▼▪†■’▀¨▄♫☆é¯♦¤▲è¸¾Ã⋅‘∞∙）↓、│（»，♪╩╚³・╦╣╔╗▬❤ïØ¹≤‡√'\n\n#Funcion para obtener todos los caracteres desconocidos entre el embedding y la lista de caracteres\ndef caracteres_desconocidos(embed, puntuacion):\n    desconocido = ''\n    for p in puntuacion:\n        if p not in embed:\n            desconocido += p\n            desconocido += ' '\n    return desconocido\n\n\nprint(\"Glove:\")#Imprimiendo los caracteres desconocidos\nprint(caracteres_desconocidos(embed_glove, mapeo_puntuacion))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#se define  el diccionario para mapear los caracteres especiales\npuntuacion = {\"‘\": \"'\", \"´\": \"'\", \"°\": \"\", \"€\": \"e\", \"—\": \"-\", \"–\": \"-\", \"’\": \"'\", \"_\": \"-\", \"`\": \"'\", '“': '\"', '”': '\"', '“': '\"', \"£\": \"e\", '∞': 'infinity', 'θ': 'theta', '÷': '/', 'α': 'alpha', '•': '.', 'à': 'a', '−': '-', 'β': 'beta', '∅': '', '³': '3', 'π': 'pi', '…': ' '}\n\n#Se define la función eliminar_caracteres para eliminar caracteres desconocidos y reemplazarlos por el correspondiente\ndef eliminar_caracteres(texto, puntuacion, mapeo):\n    for p in mapeo:\n        texto = texto.replace(p, mapeo[p])\n    \n    for p in puntuacion:\n        texto = texto.replace(p, f' {p} ')\n    \n    return texto\n\n#Eliminando caracteres especiales\nentrenamiento['question_text'] = entrenamiento['question_text'].apply(lambda x: eliminar_caracteres(x, mapeo_puntuacion, puntuacion))\nprueba['question_text'] = prueba['question_text'].apply(lambda x: eliminar_caracteres(x, mapeo_puntuacion, puntuacion))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Ahora se reconstruye el diccionario de palabras luego de los cambios\ndf = pd.concat([entrenamiento ,prueba], sort=False)\ndiccionario = construccion_diccionario(df['question_text'])\n\n#Imprimiendo las primeras 10 palabras desconocidas del glove\nprint(\"Glove: \")\npalabras_desconocidas = check_coverage(diccionario, embed_glove)\npalabras_desconocidas[:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nentrenamiento, val = train_test_split(entrenamiento, test_size=0.2, random_state=42) #Se reserva  un 10% para el conjunto de validacion\n\n#Filtrando los datos para evitar errores\nxentrenamiento = entrenamiento['question_text'].fillna('_na_').values\nxval = val['question_text'].fillna('_na_').values\nxprueba = prueba['question_text'].fillna('_na_').values\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Tokenizaremos oraciones segun el parametro MAX_FEATURES\ntokenizer = Tokenizer(num_words=MAX_FEATURES)\ntokenizer.fit_on_texts(list(xentrenamiento))\n\n#Tokenizamos el conjunto de entrenamineto, validacion y pruebas\nxentrenamiento = tokenizer.texts_to_sequences(xentrenamiento)\nxval = tokenizer.texts_to_sequences(xval)\nxprueba = tokenizer.texts_to_sequences(xprueba)\nprint(xentrenamiento[0])\n#Nos aseguraremos de que cada oracion tenga un tamaño MAXLEN\nxentrenamiento = pad_sequences(xentrenamiento, maxlen=MAXLEN)\nxval = pad_sequences(xval, maxlen=MAXLEN)\nxprueba = pad_sequences(xprueba, maxlen=MAXLEN)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Definiendo las salidas esperadas y mezclando el modelo para una mayor generalizacion\nyentrenamiento = entrenamiento['target'].values\nyval = val['target'].values\n\n#Mezclando el conjunto de datos\nnp.random.seed(42)\n\ntrn_idx = np.random.permutation(len(xentrenamiento))\nval_idx = np.random.permutation(len(xval))\n\nxentrenamiento = xentrenamiento[trn_idx]\nyentrenamiento = yentrenamiento[trn_idx]\nxval = xval[val_idx]\nyval = yval[val_idx]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Cargando la matriz glove de embeddings\nmatriz_embedding_glove = cargar_m_glove(tokenizer.word_index, embed_glove)\nprint(\"Matriz de embeddings cargada!\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class Attention(Layer):\n    def __init__(self, step_dim,\n                 W_regularizer=None, b_regularizer=None,\n                 W_constraint=None, b_constraint=None,\n                 bias=True, **kwargs):\n        self.supports_masking = True\n        self.init = initializers.get('glorot_uniform')\n\n        self.W_regularizer = regularizers.get(W_regularizer)\n        self.b_regularizer = regularizers.get(b_regularizer)\n\n        self.W_constraint = constraints.get(W_constraint)\n        self.b_constraint = constraints.get(b_constraint)\n\n        self.bias = bias\n        self.step_dim = step_dim\n        self.features_dim = 0\n        super(Attention, self).__init__(**kwargs)\n\n    def build(self, input_shape):\n        assert len(input_shape) == 3\n        shapeW=(input_shape[-1],)\n        shapeB=(input_shape[1],)\n        self.W = self.add_weight(shape=shapeW,\n                                 initializer=self.init,\n                                 name='{}_W'.format(self.name),\n                                 regularizer=self.W_regularizer,\n                                 constraint=self.W_constraint)\n        self.features_dim = input_shape[-1]\n\n        if self.bias:\n            self.b = self.add_weight(shape=shapeB,\n                                     initializer='zero',\n                                     name='{}_b'.format(self.name),\n                                     regularizer=self.b_regularizer,\n                                     constraint=self.b_constraint)\n        else:\n            self.b = None\n\n        self.built = True\n\n    def compute_mask(self, input, input_mask=None):\n        return None\n\n    def call(self, x, mask=None):\n        features_dim = self.features_dim\n        step_dim = self.step_dim\n\n        eij = K.reshape(K.dot(K.reshape(x, (-1, features_dim)),\n                        K.reshape(self.W, (features_dim, 1))), (-1, step_dim))\n\n        if self.bias:\n            eij += self.b\n\n        eij = K.tanh(eij)\n        a = K.exp(eij)\n\n        if mask is not None:\n            a *= K.cast(mask, K.floatx())\n\n        a /= K.cast(K.sum(a, axis=1, keepdims=True) + K.epsilon(), K.floatx())\n\n        a = K.expand_dims(a)\n        weighted_input = x * a\n        return K.sum(weighted_input, axis=1)\n\n    def compute_output_shape(self, input_shape):\n        return input_shape[0], self.features_dim","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def f1(y_true, y_pred):\n\n    def recall(y_true, y_pred):\n        \n        true_positives = K.sum(K.round(K.clip(y_true*y_pred, 0, 1)))\n        possible_positives = K.sum(K.round(K.clip(y_true, 0, 1)))\n        recall = true_positives/(possible_positives + K.epsilon())\n        return recall\n\n    def precision(y_true, y_pred):\n        \n        true_positives = K.sum(K.round(K.clip(y_true*y_pred, 0, 1)))\n        predicted_positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n        precision = true_positives/(predicted_positives + K.epsilon())\n        return precision\n\n    precision = precision(y_true, y_pred)\n    recall = recall(y_true, y_pred)\n\n    return 2*((precision*recall)/(precision+recall+K.epsilon()))\n\n\ndef model_lstm_att(matriz_embedding):\n    \n    inp = Input(shape=(MAXLEN,))\n    x = Embedding(MAX_FEATURES, tamano_embedding, weights=[matriz_embedding], trainable=False)(inp)\n    x = Bidirectional(LSTM(64, return_sequences=True))(x)\n    x = Bidirectional(LSTM(32, return_sequences=True))(x)\n    \n    att = Attention(MAXLEN)(x)\n    \n    y = Dense(32, activation='relu')(att)\n    y = Dropout(0.1)(y)\n    outp = Dense(1, activation='sigmoid')(y)    \n\n    model = Model(inputs=inp, outputs=outp)\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=[f1, \n                                                                        \"acc\"])\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def train_pred(model, epochs=2):\n    \n    for e in range(epochs):\n        model.fit(xentrenamiento, yentrenamiento, batch_size=512, epochs=3, validation_data=(xval, yval))\n        pred_val_y = model.predict([xval], batch_size=1024, verbose=0)\n        best_thresh = 0.5\n        best_score = 0.0\n        for thresh in np.arange(0.1, 0.501, 0.01):\n            thresh = np.round(thresh, 2)\n            score = metrics.f1_score(yval, (pred_val_y > thresh).astype(int))\n            if score > best_score:\n                best_thresh = thresh\n                best_score = score\n\n        print(\"Val F1 Score: {:.4f}\".format(best_score))\n\n    pred_test_y = model.predict([xtest], batch_size=1024, verbose=0)\n\n    return pred_val_y, pred_test_y, best_score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def model_lstm_att(embedding_matrix):\n    \n    inp = Input(shape=(MAXLEN,))\n    x = Embedding(MAX_FEATURES, tamano_embedding, weights=[matriz_embedding], trainable=False)(inp)\n    x = Bidirectional(LSTM(64, return_sequences=True))(x)\n    x = Bidirectional(LSTM(32, return_sequences=True))(x)\n    \n    att = Attention(MAXLEN)(x)\n    \n    y = Dense(32, activation='relu')(att)\n    y = Dropout(0.1)(y)\n    outp = Dense(1, activation='sigmoid')(y)    \n\n    model = Model(inputs=inp, outputs=outp)\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=[f1, \n                                                                        \"acc\"])\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def train_pred(model, epochs=2):\n    \n    for e in range(epochs):\n        model.fit(xentrenamiento, yentrenamiento, batch_size=512, epochs=3, validation_data=(xval, yval))\n        pred_val_y = model.predict([xval], batch_size=1024, verbose=0)\n        best_thresh = 0.5\n        best_score = 0.0\n        for thresh in np.arange(0.1, 0.501, 0.01):\n            thresh = np.round(thresh, 2)\n            score = metrics.f1_score(yval, (pred_val_y > thresh).astype(int))\n            if score > best_score:\n                best_thresh = thresh\n                best_score = score\n\n        print(\"Val F1 Score: {:.4f}\".format(best_score))\n\n    pred_test_y = model.predict([xprueba], batch_size=1024, verbose=0)\n\n    return pred_val_y, pred_test_y, best_score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"paragram = '../input/quora-insincere-questions-classification/embeddings/paragram_300_sl999/paragram_300_sl999.txt'\nembedding_matrix_para = cargar_m_glove(tokenizer.word_index, cargar_embedding(paragram))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"matriz_embedding = np.mean([matriz_embedding_glove, embedding_matrix_para], axis=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#creacion y entrenamiento del modelo\nmodel_lstm = model_lstm_att(matriz_embedding)\nmodel_lstm.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"outputs = []\npred_val_y, pred_test_y, best_score = train_pred(model_lstm, epochs=3)\noutputs.append([pred_val_y, pred_test_y, best_score, 'model_lstm_att only Glove'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#encontrar el mejor threshold\noutputs.sort(key=lambda x: x[2]) \nweights = [i for i in range(1, len(outputs) + 1)]\nweights = [float(i) / sum(weights) for i in weights] \n\npred_val_y = np.mean([outputs[i][0] for i in range(len(outputs))], axis = 0)\n\nthresholds = []\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    res = metrics.f1_score(yval, (pred_val_y > thresh).astype(int))\n    thresholds.append([thresh, res])\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, res))\n    \nthresholds.sort(key=lambda x: x[1], reverse=True)\nbest_thresh = thresholds[0][0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Best threshold:\", best_thresh, \"and F1 score\", thresholds[0][1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#prediciones y archivo para el submit\npred_test_y = np.mean([outputs[i][1] for i in range(len(outputs))], axis = 0)\npred_test_y = (pred_test_y > best_thresh).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub = pd.read_csv('../input/quora-insincere-questions-classification/sample_submission.csv')\nout_df = pd.DataFrame({\"qid\":sub[\"qid\"].values})\nout_df['prediction'] = pred_test_y\nout_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}