{"cells":[{"metadata":{},"cell_type":"markdown","source":"**Proyecto 3 - Computacion Emergente**\n\n*Estudiantes: Abraham Chang, Luciano Pinedo, Daniel Velasquez*\n"},{"metadata":{},"cell_type":"markdown","source":"**IMPORTANDO LIBRERIAS**"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\n\nimport pandas as pd\n\nimport operator\n\nfrom sklearn import metrics\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.model_selection import train_test_split\n\nfrom keras import backend as K, initializers, regularizers, constraints, optimizers, layers\nfrom keras.layers import Dense, Input, CuDNNLSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D, concatenate\nfrom keras.layers import Bidirectional, GlobalMaxPool1D, GlobalMaxPooling1D, GlobalAveragePooling1D\nfrom keras.layers import Input, Embedding, Dense, Conv2D, MaxPool2D, concatenate, Lambda\nfrom keras.layers import Reshape, Flatten, Concatenate, Dropout, SpatialDropout1D\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.optimizers import Adam\nfrom keras.models import Model\nfrom keras.engine.topology import Layer\n\nimport os\nprint(os.listdir(\"../input\"))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**PROCESAMIENTO DE LOS DATOS**"},{"metadata":{"trusted":true},"cell_type":"code","source":"#Revisando el tamaño del dataset\ntrain = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")\n\nprint(\"Train shape : \",train.shape)\nprint(\"Test shape : \",test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Definiendo los parametros para el procesamiento de los datos\nEMBED_SIZE = 300 # Este es el tamaño de vector de palabras\nMAX_FEATURES = 100000 # Cantidad máxima de palabras a tomar en cuenta. Si el vocabulario supera este número, Keras escoge las palabras con mayor frecuencia. \nMAXLEN = 40 # Este sera el tamaño maximo de la pregunta","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Construyendo el diccionario de palabras: Creamos un diccionario, donde el key es la palabra, y el valor es la frecuencia de esa palabra. \ndef build_vocab(texts): #reconstruiremos el diccionario varias veces durante el pre-procesamiento para ver cambios\n    sentences = texts.apply(lambda x: x.split()).values #Dividimos las oraciones en una lista de arreglos, donde cada arreglo representa una oración. Las celdas en ese arreglo son palabras.\n    vocab = {}\n    for sentence in sentences:\n        for word in sentence:\n            #Contamos cada palabra en por oración. Si existe en el vocabulario, se le suma 1 a las veces que se repite. Si no existe, se agrega al vocabulario con 1. \n            try:\n                vocab[word] += 1\n            except KeyError:\n                vocab[word] = 1\n    return vocab\n\ndf = pd.concat([train ,test], sort=False)\n\nvocab = build_vocab(df['question_text'])\nprint(\"Tamaño inicial del vocabulario:\")\nprint(len(vocab))\n#Imprimiendo los primeros 10 elementos del diccionario\nfor x in list(vocab)[0:10]:\n    print (x, vocab[x])\n    print()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Funcion para cargar la matriz de embeddings y sus indices \n\ndef cargar_embedding(file):\n    def get_coeficientes(word,*arr): \n        return word, np.asarray(arr, dtype='float32')\n    embeddings_index = dict(get_coeficientes(*o.split(\" \")) for o in open(file, encoding='latin'))\n    return embeddings_index\n\nglove = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\n\nembed_glove = cargar_embedding(glove)\n\nprint('Glove embeddings cargados!')\nlen(embed_glove)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Funcion para cargar la matriz de glove\n\ndef cargar_matriz_glove(word_index, embeddings_index):\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean, emb_std = all_embs.mean(), all_embs.std()\n    EMBED_SIZE = all_embs.shape[1]\n    \n    nb_words = min(MAX_FEATURES, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, EMBED_SIZE))\n\n    for word, i in word_index.items():\n        if i >= MAX_FEATURES:\n            continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None:\n            embedding_matrix[i] = embedding_vector\n\n    return embedding_matrix","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Funcion para evaluar el coverage entre el diccionacio y un conjunto de embedding\n\n\"\"\"\nLa funcion calcula la intersección entre el diccionario y los embeddings\nSi hay una palabra del diccionario que no este en el embedding se devuelve como desconocida. \n\"\"\"\ndef check_coverage(vocab, embeddings_index):\n    known_words = {}\n    unknown_words = {}\n    nb_known_words = 0\n    nb_unknown_words = 0\n    for word in vocab.keys():\n        try:\n            known_words[word] = embeddings_index[word]\n            nb_known_words += vocab[word]\n        except:\n            unknown_words[word] = vocab[word]\n            nb_unknown_words += vocab[word]\n            pass\n    \n    print(\"Glove\")\n    print('Se encontraron embeddings para el {:.3%} del diccionario'.format(len(known_words)/len(vocab)))\n    print('Se encontraron embeddings para el {:.3%} de todo el cuerpo de texto'.format(nb_known_words/(nb_known_words + nb_unknown_words)))\n    unknown_words = sorted(unknown_words.items(), key=operator.itemgetter(1))[::-1]\n\n    return unknown_words","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"unknown_words = check_coverage(vocab, embed_glove)\n\n#Para mejorar el modelo es util observar que palabras no estan en el diccionario\nunknown_words[:20]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Aun el porcentaje de palabras desconocidas es significativo, esto puede ser producto de que hay palabras en mayusculas, caracteres especiales o contracciones tipicas del ingles. Por ello nos enfocaremos en seguir haciendo un pre-procesamiento de la data"},{"metadata":{},"cell_type":"markdown","source":"**LLEVANDO TODO A MINUSCULAS**"},{"metadata":{"trusted":true},"cell_type":"code","source":"#Funcion para agregar minusculas al embedding\ndef agregar_minusculas(embedding, vocab):\n    count = 0\n    for word in vocab:\n        if word in embedding and word.lower() not in embedding:  \n            embedding[word.lower()] = embedding[word]\n            count += 1\n    print(f\"Added {count} words to embedding\")\n    \n#Llevando todo a minusculas\ntrain['question_text'] = train['question_text'].apply(lambda x: x.lower())\ntest['question_text'] = test['question_text'].apply(lambda x: x.lower())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Imprimiendo el Glove!\")\n#Previo\nunknown_words = check_coverage(vocab, embed_glove)\n\n#Actualizado\nagregar_minusculas(embed_glove, vocab) \nunknown_words = check_coverage(vocab, embed_glove)\n\n#Imprimiendo las primeras 10 palabras desconocidas del glove\nunknown_words[:10]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Llevar todo a minusculas no origino un gran cambio en el glove"},{"metadata":{},"cell_type":"markdown","source":"**MAPEANDO LAS CONTRACCIONES**\n\nPara mejorar la eficiencia del modelo intentaremos mejorar los datos, esta vez mapeando las contracciones con ayuda de un diccionario que ha sido util para otras personas en esta competencia.\n\nDiccionario de contracciones: https://www.kaggle.com/c/quora-insincere-questions-classification/discussion/77758\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"#Definiendo el diccionario para mapear las contracciones segun el enlace anterior\ncontraction_mapping = {\"ain't\": \"is not\", \"aren't\": \"are not\",\"can't\": \"cannot\", \"'cause\": \"because\", \"could've\": \"could have\", \"couldn't\": \"could not\", \"didn't\": \"did not\",  \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\", \"he'd\": \"he would\",\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\", \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\",  \"I'd\": \"I would\", \"I'd've\": \"I would have\", \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"I've\": \"I have\", \"i'd\": \"i would\", \"i'd've\": \"i would have\", \"i'll\": \"i will\",  \"i'll've\": \"i will have\",\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'd've\": \"it would have\", \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\", \"ma'am\": \"madam\", \"mayn't\": \"may not\", \"might've\": \"might have\",\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \"must've\": \"must have\", \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \"needn't\": \"need not\", \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\", \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\", \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\", \"she'll\": \"she will\", \"she'll've\": \"she will have\", \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\", \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \"there'd've\": \"there would have\", \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\", \"they'll\": \"they will\", \"they'll've\": \"they will have\", \"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\", \"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\", \"we'll've\": \"we will have\", \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\", \"what're\": \"what are\",  \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\", \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\"you'd\": \"you would\", \"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\", 'colour': 'color', 'centre': 'center', 'favourite': 'favorite', 'travelling': 'traveling', 'counselling': 'counseling', 'theatre': 'theater', 'cancelled': 'canceled', 'labour': 'labor', 'organisation': 'organization', 'wwii': 'world war 2', 'citicise': 'criticize', 'youtu ': 'youtube ', 'Qoura': 'Quora', 'sallary': 'salary', 'Whta': 'What', 'narcisist': 'narcissist', 'howdo': 'how do', 'whatare': 'what are', 'howcan': 'how can', 'howmuch': 'how much', 'howmany': 'how many', 'whydo': 'why do', 'doI': 'do I', 'theBest': 'the best', 'howdoes': 'how does', 'mastrubation': 'masturbation', 'mastrubate': 'masturbate', \"mastrubating\": 'masturbating', 'pennis': 'penis', 'Etherium': 'Ethereum', 'narcissit': 'narcissist', 'bigdata': 'big data', '2k17': '2017', '2k18': '2018', 'qouta': 'quota', 'exboyfriend': 'ex boyfriend', 'airhostess': 'air hostess', \"whst\": 'what', 'watsapp': 'whatsapp', 'demonitisation': 'demonetization', 'demonitization': 'demonetization', 'demonetisation': 'demonetization'}\n\n#Tamaño del diccionario de mapeo de contracciones\nlen(contraction_mapping)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Funcion para mapear las contracciones en ingles!\ndef quitar_contracciones(text, mapping):\n    specials = [\"’\", \"‘\", \"´\", \"`\"]\n    for s in specials:\n        text = text.replace(s, \"'\")\n    text = ' '.join([mapping[t] if t in mapping else t for t in text.split(\" \")])\n    return text\n\n#Eliminando las contracciones!\ntrain['question_text'] = train['question_text'].apply(lambda x: quitar_contracciones(x, contraction_mapping))\ntest['question_text'] = test['question_text'].apply(lambda x: quitar_contracciones(x, contraction_mapping))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Reconstruyendo el diccionario de palabras luego de los cambios\ndf = pd.concat([train ,test], sort=False)\nvocab = build_vocab(df['question_text'])\n\n#Imprimiendo las primeras 10 palabras desconocidas del glove\nprint(\"Glove: \")\nunknown_words = check_coverage(vocab, embed_glove)\nunknown_words[:10]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Tras llevar todo a minusculas y eliminar las contracciones, no hay cambios cosiderables y aun es posible mejorar!"},{"metadata":{},"cell_type":"markdown","source":"**ELIMINANDO CARACTERES ESPECIALES**"},{"metadata":{"trusted":true},"cell_type":"code","source":"#Definiendo los caracteres especiales...\npunct_mapping = \"/-'?!.,#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~\" + '\"\"“”’' + '∞θ÷α•à−β∅³π‘₹´°£€\\×™√²—–&'\npunct_mapping += '©^®` <→°€™› ♥←×§″′Â█½à…“★”–●â►−¢²¬░¶↑±¿▾═¦║―¥▓—‹─▒：¼⊕▼▪†■’▀¨▄♫☆é¯♦¤▲è¸¾Ã⋅‘∞∙）↓、│（»，♪╩╚³・╦╣╔╗▬❤ïØ¹≤‡√'\n\n#Funcion para obtener todos los caracteres desconocidos entre el embedding y la lista de caracteres\ndef caracteres_desconocidos(embed, punct):\n    unknown = ''\n    for p in punct:\n        if p not in embed:\n            unknown += p\n            unknown += ' '\n    return unknown\n\n#Imprimiendo los caracteres desconocidos!\nprint(\"Glove:\")\nprint(caracteres_desconocidos(embed_glove, punct_mapping))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Definiendo el diccionario para mapear los caracteres especiales\npuncts = {\"‘\": \"'\", \"´\": \"'\", \"°\": \"\", \"€\": \"e\", \"—\": \"-\", \"–\": \"-\", \"’\": \"'\", \"_\": \"-\", \"`\": \"'\", '“': '\"', '”': '\"', '“': '\"', \"£\": \"e\", '∞': 'infinity', 'θ': 'theta', '÷': '/', 'α': 'alpha', '•': '.', 'à': 'a', '−': '-', 'β': 'beta', '∅': '', '³': '3', 'π': 'pi', '…': ' '}\n\n#Funcion para eliminar caracteres desconocidos y reemplazarlos por el correspondiente\ndef eliminar_caracteres(text, punct, mapping):\n    for p in mapping:\n        text = text.replace(p, mapping[p])\n    \n    for p in punct:\n        text = text.replace(p, f' {p} ')\n    \n    return text\n\n#Eliminando caracteres especiales\ntrain['question_text'] = train['question_text'].apply(lambda x: eliminar_caracteres(x, punct_mapping, puncts))\ntest['question_text'] = test['question_text'].apply(lambda x: eliminar_caracteres(x, punct_mapping, puncts))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Reconstruyendo el diccionario de palabras luego de los cambios\ndf = pd.concat([train ,test], sort=False)\nvocab = build_vocab(df['question_text'])\n\n#Imprimiendo las primeras 10 palabras desconocidas del glove\nprint(\"Glove: \")\nunknown_words = check_coverage(vocab, embed_glove)\nunknown_words[:10]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"En esta ocasion, tras eliminar todos los caracteres especiales desconocidos por el Glove o sustituirlos por caracteres soportados, obtenemos una gran mejora, esta vez contamos con embeddings para el 99% de todo el conjunto"},{"metadata":{},"cell_type":"markdown","source":"**DEFINIENDO UN CONJUNTO DE VALIDACION**\n\nReservaremos un grupo de datos para hacer pruebas y validaciones del modelo antes de hacer el submission a kaggle"},{"metadata":{"trusted":true},"cell_type":"code","source":"#Reservando un 10% para el conjunto de validacion\ntrain, val = train_test_split(train, test_size=0.2, random_state=42)\n\n#Filtrando los datos para evitar errores\nxtrain = train['question_text'].fillna('_na_').values\nxval = val['question_text'].fillna('_na_').values\nxtest = test['question_text'].fillna('_na_').values","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**TOKENIZANDO ORACIONES**"},{"metadata":{"trusted":true},"cell_type":"code","source":"#Tokenizaremos oraciones segun el parametro MAX_FEATURES que se definio antes, es decir, 10000\ntokenizer = Tokenizer(num_words=MAX_FEATURES)\ntokenizer.fit_on_texts(list(xtrain))\n\n#Tokenizamos el conjunto de entrenamineto, validacion y pruebas\nxtrain = tokenizer.texts_to_sequences(xtrain)\nxval = tokenizer.texts_to_sequences(xval)\nxtest = tokenizer.texts_to_sequences(xtest)\nprint(xtrain[0])\n#Nos aseguraremos de que cada oracion tenga un tamaño MAXLEN, definido anteriormente, es decir, 40\nxtrain = pad_sequences(xtrain, maxlen=MAXLEN)\nxval = pad_sequences(xval, maxlen=MAXLEN)\nxtest = pad_sequences(xtest, maxlen=MAXLEN)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**MEZCLANDO EL CONJUNTO DE ENTRENAMIENTO**\n\nMezclar el conjunto de entrenamiento puede ayudarnos a obtener mejores resultados"},{"metadata":{"trusted":true},"cell_type":"code","source":"#Definiendo las salidas esperadas y mezclando el modelo para una mayor generalizacion\nytrain = train['target'].values\nyval = val['target'].values\n\n#Mezclando el conjunto de datos\nnp.random.seed(42)\n\ntrn_idx = np.random.permutation(len(xtrain))\nval_idx = np.random.permutation(len(xval))\n\nxtrain = xtrain[trn_idx]\nytrain = ytrain[trn_idx]\nxval = xval[val_idx]\nyval = yval[val_idx]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Cargando la matriz glove de embeddings\nembedding_matrix_glove = cargar_matriz_glove(tokenizer.word_index, embed_glove)\nprint(\"Matriz de embeddings cargada!\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**CONSTRUYENDO EL MODELO**\n\n"},{"metadata":{},"cell_type":"markdown","source":"**CAPA ATTENTION**\n\nDefiniremos una capa Attention que permitira mejorar el modelo. Esta capa usa las salidas de todas las hidden layers y calcula cuanta importancia debe darle a cada una. Con los resultados de esta capa crearemos una matriz que sirva de contexto en cada timestep."},{"metadata":{"trusted":true},"cell_type":"code","source":"class Attention(Layer):\n    def __init__(self, step_dim,\n                 W_regularizer=None, b_regularizer=None,\n                 W_constraint=None, b_constraint=None,\n                 bias=True, **kwargs):\n        self.supports_masking = True\n        self.init = initializers.get('glorot_uniform')\n\n        self.W_regularizer = regularizers.get(W_regularizer)\n        self.b_regularizer = regularizers.get(b_regularizer)\n\n        self.W_constraint = constraints.get(W_constraint)\n        self.b_constraint = constraints.get(b_constraint)\n\n        self.bias = bias\n        self.step_dim = step_dim\n        self.features_dim = 0\n        super(Attention, self).__init__(**kwargs)\n\n    def build(self, input_shape):\n        assert len(input_shape) == 3\n\n        self.W = self.add_weight((input_shape[-1],),\n                                 initializer=self.init,\n                                 name='{}_W'.format(self.name),\n                                 regularizer=self.W_regularizer,\n                                 constraint=self.W_constraint)\n        self.features_dim = input_shape[-1]\n\n        if self.bias:\n            self.b = self.add_weight((input_shape[1],),\n                                     initializer='zero',\n                                     name='{}_b'.format(self.name),\n                                     regularizer=self.b_regularizer,\n                                     constraint=self.b_constraint)\n        else:\n            self.b = None\n\n        self.built = True\n\n    def compute_mask(self, input, input_mask=None):\n        return None\n\n    def call(self, x, mask=None):\n        features_dim = self.features_dim\n        step_dim = self.step_dim\n\n        eij = K.reshape(K.dot(K.reshape(x, (-1, features_dim)),\n                        K.reshape(self.W, (features_dim, 1))), (-1, step_dim))\n\n        if self.bias:\n            eij += self.b\n\n        eij = K.tanh(eij)\n        a = K.exp(eij)\n\n        if mask is not None:\n            a *= K.cast(mask, K.floatx())\n\n        a /= K.cast(K.sum(a, axis=1, keepdims=True) + K.epsilon(), K.floatx())\n\n        a = K.expand_dims(a)\n        weighted_input = x * a\n        return K.sum(weighted_input, axis=1)\n\n    def compute_output_shape(self, input_shape):\n        return input_shape[0], self.features_dim","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**F1 SCORE**\n\nUsaremos la metrica F1, esta es una metrica de uso comun en algoritmos de clasificacion. Funciona como un promedio ponderado entre la precision del modelo y su memoria, la puntuacion es excelente para valores cercanos a 1 y deficiente para valores cercanos a 0"},{"metadata":{"trusted":true},"cell_type":"code","source":"def f1(y_true, y_pred):\n\n    def recall(y_true, y_pred):\n        \n        true_positives = K.sum(K.round(K.clip(y_true*y_pred, 0, 1)))\n        possible_positives = K.sum(K.round(K.clip(y_true, 0, 1)))\n        recall = true_positives/(possible_positives + K.epsilon())\n        return recall\n\n    def precision(y_true, y_pred):\n        \n        true_positives = K.sum(K.round(K.clip(y_true*y_pred, 0, 1)))\n        predicted_positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n        precision = true_positives/(predicted_positives + K.epsilon())\n        return precision\n\n    precision = precision(y_true, y_pred)\n    recall = recall(y_true, y_pred)\n\n    return 2*((precision*recall)/(precision+recall+K.epsilon()))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**MODELO LSTM + ATTENTION**"},{"metadata":{"trusted":true},"cell_type":"code","source":"def model_lstm_att(embedding_matrix):\n    \n    inp = Input(shape=(MAXLEN,))\n    x = Embedding(MAX_FEATURES, EMBED_SIZE, weights=[embedding_matrix], trainable=False)(inp)\n    x = Bidirectional(CuDNNLSTM(64, return_sequences=True))(x)\n    x = Bidirectional(CuDNNLSTM(32, return_sequences=True))(x)\n    \n    att = Attention(MAXLEN)(x)\n    \n    y = Dense(32, activation='relu')(att)\n    y = Dropout(0.1)(y)\n    outp = Dense(1, activation='sigmoid')(y)    \n\n    model = Model(inputs=inp, outputs=outp)\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=[f1, \n                                                                        \"acc\"])\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Entrenamiento del modelo!\ndef train_pred(model, epochs=2):\n    \n    for e in range(epochs):\n        model.fit(xtrain, ytrain, batch_size=512, epochs=3, validation_data=(xval, yval))\n        pred_val_y = model.predict([xval], batch_size=1024, verbose=0)\n        best_thresh = 0.5\n        best_score = 0.0\n        for thresh in np.arange(0.1, 0.501, 0.01):\n            thresh = np.round(thresh, 2)\n            score = metrics.f1_score(yval, (pred_val_y > thresh).astype(int))\n            if score > best_score:\n                best_thresh = thresh\n                best_score = score\n\n        print(\"Val F1 Score: {:.4f}\".format(best_score))\n\n    pred_test_y = model.predict([xtest], batch_size=1024, verbose=0)\n\n    return pred_val_y, pred_test_y, best_score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"paragram = '../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt'\nembedding_matrix_para = cargar_matriz_glove(tokenizer.word_index, cargar_embedding(paragram))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Se utiliza el promedio de los embeddings glove y paragram"},{"metadata":{"trusted":true},"cell_type":"code","source":"embedding_matrix = np.mean([embedding_matrix_glove, embedding_matrix_para], axis=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#creacion y entrenamiento del modelo\nmodel_lstm = model_lstm_att(embedding_matrix)\nmodel_lstm.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"outputs = []\npred_val_y, pred_test_y, best_score = train_pred(model_lstm, epochs=3)\noutputs.append([pred_val_y, pred_test_y, best_score, 'model_lstm_att only Glove'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#find best threshold\noutputs.sort(key=lambda x: x[2]) \nweights = [i for i in range(1, len(outputs) + 1)]\nweights = [float(i) / sum(weights) for i in weights] \n\npred_val_y = np.mean([outputs[i][0] for i in range(len(outputs))], axis = 0)\n\nthresholds = []\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    res = metrics.f1_score(yval, (pred_val_y > thresh).astype(int))\n    thresholds.append([thresh, res])\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, res))\n    \nthresholds.sort(key=lambda x: x[1], reverse=True)\nbest_thresh = thresholds[0][0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Best threshold:\", best_thresh, \"and F1 score\", thresholds[0][1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#prediciones y archivo para el submit\npred_test_y = np.mean([outputs[i][1] for i in range(len(outputs))], axis = 0)\npred_test_y = (pred_test_y > best_thresh).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub = pd.read_csv('../input/sample_submission.csv')\nout_df = pd.DataFrame({\"qid\":sub[\"qid\"].values})\nout_df['prediction'] = pred_test_y\nout_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Bibliografía\n\nFunción attention tomada de https://www.kaggle.com/kiraplenkin/model-lstm-attention/notebook"}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}