{"cells":[{"metadata":{},"cell_type":"markdown","source":"### Librerias"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport time\nimport numpy as np\nimport pandas as pd \nfrom tqdm import tqdm\nfrom keras.engine.topology import Layer\nimport math\nimport operator \nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.preprocessing import StandardScaler\nfrom keras import regularizers\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D, TimeDistributed, CuDNNLSTM,Conv2D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D, GlobalAveragePooling1D, concatenate, Flatten, Reshape, AveragePooling2D, Average\nfrom keras.models import Model\nfrom keras.layers import Wrapper\nfrom keras.models import Model\nfrom keras.layers import Dense, Embedding, Bidirectional, CuDNNGRU, GlobalAveragePooling1D, GlobalMaxPooling1D, concatenate, Input, Dropout\nfrom keras.optimizers import Adam\nimport keras.backend as K\nimport matplotlib as plt\nfrom keras.callbacks import ModelCheckpoint, ReduceLROnPlateau\nfrom keras import initializers, regularizers, constraints, optimizers, layers\ntqdm.pandas()\nimport pandas as pd\nimport numpy as np\nimport operator \nimport re\nimport gc\nimport keras\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\n\nsns.set_style('whitegrid')\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Adquisicion de la data"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(\"../input/train.csv\")\ntest_df = pd.read_csv(\"../input/test.csv\")\nprint(\"Train shape : \",train_df.shape)\nprint(\"Test shape : \",test_df.shape)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Dividimos data del train para la validacion"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df, val_df = train_test_split(train_df, test_size=0.1, random_state=2)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Rellenamos los valores faltantes con \"na\" (not available)"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_X = train_df[\"question_text\"].fillna(\"_na_\").values\nval_X = val_df[\"question_text\"].fillna(\"_na_\").values\ntest_X = test_df[\"question_text\"].fillna(\"_na_\").values","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Definimos algunos Hiperparametros"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Tamaño de cada vector de palabras\nembed_size = 300 \n# Cantidad de palabras unicas a usar \nmax_features = 100000 \n# Numero maximo de palabras a usar en una pregunta\nmaxlen = 70 ","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Tokenizamos la columna de dexto y las convertimos a un vector de secuencias."},{"metadata":{"trusted":true},"cell_type":"code","source":"tokenizer = Tokenizer(num_words=max_features)\ntokenizer.fit_on_texts(list(train_X))\ntrain_X = tokenizer.texts_to_sequences(train_X)\nval_X = tokenizer.texts_to_sequences(val_X)\ntest_X = tokenizer.texts_to_sequences(test_X)\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Se hace el padding de las secuencias.\nEl pad_sequences permite tener una longitud definida para las frases, si las secuencias son mayores en\nlongitud que maxlen se truncan para que se ajuste a maxlen, y si las secuencias son menores que el maxlen se rellenan con zeros (0) para alcanzar el maxlen"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_X = pad_sequences(train_X, maxlen=maxlen)\nval_X = pad_sequences(val_X, maxlen=maxlen)\ntest_X = pad_sequences(test_X, maxlen=maxlen)\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Guardar valores objetivos (target)\nPara saber cuales son toxicas (1) y cueles no (0)"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_y = train_df['target'].values\nval_y = val_df['target'].values","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Pre-Procesamiento de la data"},{"metadata":{},"cell_type":"markdown","source":"### Cargar embeddings preentrenados"},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_embed(file):\n    def get_coefs(word,*arr): \n        return word, np.asarray(arr, dtype='float32')\n    \n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(file, encoding='latin'))\n    \n    return embeddings_index","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Se escoge GloVe como Embedding de palabras pre-entrenadas\nPara no sobre ocupar el espacio, solo escogemos de los 3 Embeddings disponibles, 1, el escogido fue GloVe"},{"metadata":{"trusted":true},"cell_type":"code","source":"glove = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\nembed_glove = load_embed(glove)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Vocabulario\nSe realiza la funcion para revisar la cantidad de veces que se encuentran las palabras en el texto."},{"metadata":{"trusted":true},"cell_type":"code","source":"#Construyendo el diccionario de palabras\ndef build_vocab(texts): #reconstruiremos el diccionario varias veces durante el pre-procesamiento \n#para ver cambios\n    sentences = texts.apply(lambda x: x.split()).values\n    vocab = {}\n    for sentence in sentences:\n        for word in sentence:\n            try:\n                vocab[word] += 1\n            except KeyError:\n                vocab[word] = 1\n    return vocab\n\n\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Palabras desconocidas\nSe realiza una funcion para revisar del embedding y el vocabulario de entrenamiento, a ver si se encuentran palabras desconocidas."},{"metadata":{"trusted":true},"cell_type":"code","source":"def check_coverage(vocab, embeddings_index):\n    known_words = {}\n    unknown_words = {}\n    nb_known_words = 0\n    nb_unknown_words = 0\n    for word in vocab.keys():\n        try:\n            known_words[word] = embeddings_index[word]\n            nb_known_words += vocab[word]\n        except:\n            unknown_words[word] = vocab[word]\n            nb_unknown_words += vocab[word]\n            pass\n\n    print('Se encontraron embeddings para {:.3%} of vocab'.format(len(known_words) / len(vocab)))\n    unknown_words = sorted(unknown_words.items(), key=operator.itemgetter(1))[::-1]\n    \n    return unknown_words","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Creacion del vocabulario con las palabras del dataset"},{"metadata":{"trusted":true},"cell_type":"code","source":"combined_df = pd.concat([train_df ,test_df]) \nvocab = build_vocab(combined_df['question_text'])\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Pasar todo a minusculas\nAunque se pueda perder algo de informacion al hacer esto, es necesario para no trabajar sobre palabras que tengan solo mayusuclas como si fueran palabras diferentes. "},{"metadata":{"trusted":true},"cell_type":"code","source":"combined_df['question_text'] = combined_df['question_text'].apply(lambda x: x.lower())\n\ndef add_lower(embedding, vocab):\n    count = 0\n    for word in vocab:\n        if word in embedding and word.lower() not in embedding:  \n            embedding[word.lower()] = embedding[word]\n            count += 1\n    print(\"Anadidas {count} palabras al embedding\")\n    \nprint(\"Glove : \")\noov_glove = check_coverage(vocab, embed_glove)\nadd_lower(embed_glove, vocab)\noov_glove = check_coverage(vocab, embed_glove)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Eliminacion de caracteres especiales,contracciones y signos de puntuación\nDebido a que no aportan mucho al modelo para la prediccion. "},{"metadata":{},"cell_type":"markdown","source":"#### Mapping Contracciones"},{"metadata":{"trusted":true},"cell_type":"code","source":"contraction_mapping = {\"ain't\": \"is not\",\n                       \"aren't\": \"are not\",\n                       \"can't\": \"cannot\",\n                       \"'cause\": \"because\",\n                       \"could've\": \"could have\",\n                       \"couldn't\": \"could not\",\n                       \"didn't\": \"did not\",\n                       \"doesn't\": \"does not\",\n                       \"don't\": \"do not\",\n                       \"hadn't\": \"had not\",\n                       \"hasn't\": \"has not\",\n                       \"haven't\": \"have not\",\n                       \"he'd\": \"he would\",\n                       \"he'll\": \"he will\",\n                       \"he's\": \"he is\",\n                       \"how'd\": \"how did\",\n                       \"how'd'y\": \"how do you\",\n                       \"how'll\": \"how will\",\n                       \"how's\": \"how is\",                       \n                       \"I'd\": \"I would\",\n                       \"I'd've\": \"I would have\",\n                       \"I'll\": \"I will\",\n                       \"I'll've\": \"I will have\",\n                       \"I'm\": \"I am\",\n                       \"I've\": \"I have\",\n                       \"i'd\": \"i would\",\n                       \"i'd've\": \"i would have\",\n                       \"i'll\": \"i will\",\n                       \"i'll've\": \"i will have\",\n                       \"i'm\": \"i am\",\n                       \"i've\": \"i have\",\n                       \"isn't\": \"is not\",\n                       \"it'd\": \"it would\",\n                       \"it'd've\": \"it would have\",\n                       \"it'll\": \"it will\",\n                       \"it'll've\": \"it will have\",\n                       \"it's\": \"it is\",\n                       \"let's\": \"let us\",\n                       \"ma'am\": \"madam\",\n                       \"mayn't\": \"may not\",                       \n                       \"might've\": \"might have\",\n                       \"mightn't\": \"might not\",\n                       \"mightn't've\": \"might not have\",\n                       \"must've\": \"must have\",\n                       \"mustn't\": \"must not\",\n                       \"mustn't've\": \"must not have\",\n                       \"needn't\": \"need not\", \n                       \"needn't've\": \"need not have\",\n                       \"o'clock\": \"of the clock\", \n                       \"oughtn't\": \"ought not\", \n                       \"oughtn't've\": \"ought not have\",\n                       \"shan't\": \"shall not\",                       \n                       \"sha'n't\": \"shall not\",\n                       \"shan't've\": \"shall not have\",\n                       \"she'd\": \"she would\",\n                       \"she'd've\": \"she would have\",\n                       \"she'll\": \"she will\",\n                       \"she'll've\": \"she will have\",                       \n                       \"she's\": \"she is\",\n                       \"should've\": \"should have\",\n                       \"shouldn't\": \"should not\",\n                       \"shouldn't've\": \"should not have\",\n                       \"so've\": \"so have\",\"so's\": \"so as\",                       \n                       \"this's\": \"this is\",\n                       \"that'd\": \"that would\",\n                       \"that'd've\": \"that would have\",\n                       \"that's\": \"that is\",\n                       \"there'd\": \"there would\",\n                       \"there'd've\": \"there would have\",                       \n                       \"there's\": \"there is\",\n                       \"here's\": \"here is\",\n                       \"they'd\": \"they would\",\n                       \"they'd've\": \"they would have\",\n                       \"they'll\": \"they will\",\n                       \"they'll've\": \"they will have\",                       \n                       \"they're\": \"they are\",\n                       \"they've\": \"they have\",\n                       \"to've\": \"to have\",\n                       \"wasn't\": \"was not\",\n                       \"we'd\": \"we would\",\n                       \"we'd've\": \"we would have\",\n                       \"we'll\": \"we will\",\n                       \"we'll've\": \"we will have\",\n                       \"we're\": \"we are\",\n                       \"we've\": \"we have\",\n                       \"weren't\": \"were not\",\n                       \"what'll\": \"what will\",\n                       \"what'll've\": \"what will have\",\n                       \"what're\": \"what are\",\n                       \"what's\": \"what is\", \n                       \"what've\": \"what have\",\n                       \"when's\": \"when is\",\n                       \"when've\": \"when have\",\n                       \"where'd\": \"where did\",\n                       \"where's\": \"where is\",\n                       \"where've\": \"where have\",\n                       \"who'll\": \"who will\",\n                       \"who'll've\": \"who will have\",\n                       \"who's\": \"who is\",\n                       \"who've\": \"who have\",\n                       \"why's\": \"why is\", \n                       \"why've\": \"why have\",\n                       \"will've\": \"will have\",\n                       \"won't\": \"will not\",\n                       \"won't've\": \"will not have\",\n                       \"would've\": \"would have\",\n                       \"wouldn't\": \"would not\",\n                       \"wouldn't've\": \"would not have\",\n                       \"y'all\": \"you all\",\n                       \"y'all'd\": \"you all would\",\n                       \"y'all'd've\": \"you all would have\",\n                       \"y'all're\": \"you all are\",\n                       \"y'all've\": \"you all have\",\n                       \"you'd\": \"you would\",\n                       \"you'd've\": \"you would have\",\n                       \"you'll\": \"you will\",\n                       \"you'll've\": \"you will have\",\n                       \"you're\": \"you are\",\n                       \"you've\": \"you have\"}\n\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Contracciones"},{"metadata":{"trusted":true},"cell_type":"code","source":"def known_contractions(embed):\n    known = []\n    for contract in contraction_mapping:\n        if contract in embed:\n            known.append(contract)\n    return known","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def clean_contractions(text, mapping):\n    specials = [\"’\", \"‘\", \"´\", \"`\"]\n    for s in specials:\n        text = text.replace(s, \"'\")\n    text = ' '.join([mapping[t] if t in mapping else t for t in text.split(\" \")])\n    return text","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"combined_df['question_text'] = combined_df['question_text'].apply(lambda x: clean_contractions(x, contraction_mapping))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Caracteres especiales"},{"metadata":{"trusted":true},"cell_type":"code","source":"carc_esp = \"/-'?!.,#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~\" + '\"\"“”’' + '∞θ÷α•à−β∅³π‘₹´°£€\\×™√²—–&'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def unknown_punct(embed, carc_esp):\n    unknown = ''\n    for c in carc_esp:\n        if c not in embed:\n            unknown += c\n            unknown += ' '\n    return unknown","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Glove :\")\nprint(unknown_punct(embed_glove, carc_esp))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Mapping de Caracteres especiales (puntuacion y caracteres)"},{"metadata":{"trusted":true},"cell_type":"code","source":"carc_esp_mapping = {\"‘\": \"'\",\n                    \"₹\": \"e\",\n                    \"´\": \"'\",\n                    \"°\": \"\",\n                    \"€\": \"e\",\n                    \"™\": \"tm\",\n                    \"√\": \" sqrt \",\n                    \"×\": \"x\", \n                    \"²\": \"2\",\n                    \"—\": \"-\",\n                    \"–\": \"-\",\n                    \"’\": \"'\",\n                    \"_\": \"-\",\n                    \"`\": \"'\",\n                    '“': '\"', \n                    '”': '\"', \n                    '“': '\"',\n                    \"£\": \"e\",\n                    '∞': 'infinity',\n                    'θ': 'theta',\n                    '÷': '/',\n                    'α': 'alpha',\n                    '•': '.', \n                    'à': 'a', \n                    '−': '-', \n                    'β': 'beta',\n                    '∅': '', \n                    '³': '3',\n                    'π': 'pi'}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def clean_special_chars(text, carc_esp, mapping):\n    for c in mapping:\n        text = text.replace(c, mapping[c])\n    \n    for c in carc_esp:\n        text = text.replace(c, f' {c} ')\n    \n    specials = {'\\u200b': ' ', '…': ' ... ', '\\ufeff': '', 'करना': '', 'है': ''}  \n    for s in specials:\n        text = text.replace(s, specials[s])\n    \n    return text","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"combined_df['question_text'] = combined_df['question_text'].apply(lambda x: clean_special_chars(x, carc_esp, carc_esp_mapping))\n\nvocab = build_vocab(combined_df['question_text'])\nprint(\"Glove : \")\noov_glove = check_coverage(vocab, embed_glove)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Mapping de Errores de pronunciacion"},{"metadata":{"trusted":true},"cell_type":"code","source":"mispell_dict = {'advanatges': 'advantages', \n                'irrationaol': 'irrational' ,\n                'defferences': 'differences',\n                'lamboghini':'lamborghini',\n                'hypothical':'hypothetical',\n                'colour': 'color', \n                'centre': 'center', \n                'favourite': 'favorite',\n                'travelling': 'traveling',\n                'counselling': 'counseling', \n                'theatre': 'theater',\n                'cancelled': 'canceled', \n                'labour': 'labor',\n                'organisation': 'organization',\n                'wwii': 'world war 2',\n                'citicise': 'criticize', \n                'youtu ': 'youtube ', \n                'Qoura': 'Quora', \n                'sallary': 'salary',\n                'Whta': 'What', \n                'narcisist': 'narcissist',\n                'howdo': 'how do',\n                'whatare': 'what are', \n                'howcan': 'how can',\n                'howmuch': 'how much',\n                'howmany': 'how many',\n                'whydo': 'why do',\n                'doI': 'do I',\n                'theBest': 'the best',\n                'howdoes': 'how does',\n                'mastrubation': 'masturbation',\n                'mastrubate': 'masturbate',\n                \"mastrubating\": 'masturbating',\n                'pennis': 'penis',\n                'Etherium': 'Ethereum',\n                'narcissit': 'narcissist',\n                'bigdata': 'big data',\n                '2k17': '2017',\n                '2k18': '2018',\n                'qouta': 'quota',\n                'exboyfriend': 'ex boyfriend',\n                'airhostess': 'air hostess',\n                \"whst\": 'what',\n                'watsapp': 'whatsapp', \n                'demonitisation': 'demonetization',\n                'demonitization': 'demonetization',\n                'demonetisation': 'demonetization',\n                'pokémon': 'pokemon'}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def correct_spelling(x, dic):\n    for word in dic.keys():\n        x = x.replace(word, dic[word])\n    return x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"combined_df['question_text'] = combined_df['question_text'].apply(lambda x: correct_spelling(x, mispell_dict))\nvocab = build_vocab(combined_df['question_text'])\nprint(\"Glove : \")\noov_glove = check_coverage(vocab, embed_glove)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Eliminar numeros"},{"metadata":{"trusted":true},"cell_type":"code","source":"def clean_numbers(x):\n\n    x = re.sub('[0-9]{5,}', ' number ', x)\n    x = re.sub('[0-9]{4}', ' number ', x)\n    x = re.sub('[0-9]{3}', ' number ', x)\n    x = re.sub('[0-9]{2}', ' number ', x)\n    return x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"combined_df['question_text'] = combined_df['question_text'].apply(lambda x: clean_numbers(x))\nvocab = build_vocab(combined_df['question_text'])\nprint(\"Glove : \")\noov_glove = check_coverage(vocab, embed_glove)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Aplicar los cambios al train y al test \nDebido a que antes se hiso solo al combinado de ambos, vamos a aplicar los cambios al train y al test respectivamente en lo que le toca a cada uno"},{"metadata":{"trusted":true},"cell_type":"code","source":"# TRAIN\n\n# Pasar a minusculas\ntrain_df['treated_question'] = train_df['question_text'].apply(lambda x: x.lower())\n# Eliminar Contracciones\ntrain_df['treated_question'] = train_df['treated_question'].apply(lambda x: clean_contractions(x, contraction_mapping))\n# Eliminar Caracteres Especiales\ntrain_df['treated_question'] = train_df['treated_question'].apply(lambda x: clean_special_chars(x, carc_esp, carc_esp_mapping))\n# Errores de Spelling\ntrain_df['treated_question'] = train_df['treated_question'].apply(lambda x: correct_spelling(x, mispell_dict))\n#Eliminar Numeros\ntrain_df['treated_question'] = train_df['treated_question'].apply(lambda x: clean_numbers(x))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#TEST\n# Pasar a minusculas\ntest_df['treated_question'] = test_df['question_text'].apply(lambda x: x.lower())\n# Eliminar Contracciones\ntest_df['treated_question'] = test_df['treated_question'].apply(lambda x: clean_contractions(x, contraction_mapping))\n# Eliminar Caracteres Especiales\ntest_df['treated_question'] = test_df['treated_question'].apply(lambda x: clean_special_chars(x, carc_esp, carc_esp_mapping))\n# Errores de Spelling\ntest_df['treated_question'] = test_df['treated_question'].apply(lambda x: correct_spelling(x, mispell_dict))\n#Eliminar Numeros\ntest_df['treated_question'] = test_df['treated_question'].apply(lambda x: clean_numbers(x))\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Fin de pre-procesamiento de data"},{"metadata":{},"cell_type":"markdown","source":"## Matriz de Embeddings"},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_glove_matrix(word_index, embeddings_index):\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean, emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n    \n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n\n    for word, i in word_index.items():\n        if i >= max_features:\n            continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None:\n            embedding_matrix[i] = embedding_vector\n\n    return embedding_matrix\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"np.random.seed(2)\n\ntrain_idx = np.random.permutation(len(train_X))\nval_idx = np.random.permutation(len(val_X))\n\ntrain_X = train_X[train_idx]\ntrain_y = train_y[train_idx]\nval_X = val_X[val_idx]\nval_y = val_y[val_idx]\n\nembedding_matrix_glove = load_glove_matrix(tokenizer.word_index, embed_glove)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Capa Attention"},{"metadata":{"trusted":true},"cell_type":"code","source":"class Attention(Layer):\n    def __init__(self, step_dim,\n                 W_regularizer=None, b_regularizer=None,\n                 W_constraint=None, b_constraint=None,\n                 bias=True, **kwargs):\n        self.supports_masking = True\n        self.init = initializers.get('glorot_uniform')\n\n        self.W_regularizer = regularizers.get(W_regularizer)\n        self.b_regularizer = regularizers.get(b_regularizer)\n\n        self.W_constraint = constraints.get(W_constraint)\n        self.b_constraint = constraints.get(b_constraint)\n\n        self.bias = bias\n        self.step_dim = step_dim\n        self.features_dim = 0\n        super(Attention, self).__init__(**kwargs)\n\n    def build(self, input_shape):\n        assert len(input_shape) == 3\n\n        self.W = self.add_weight((input_shape[-1],),\n                                 initializer=self.init,\n                                 name='{}_W'.format(self.name),\n                                 regularizer=self.W_regularizer,\n                                 constraint=self.W_constraint)\n        self.features_dim = input_shape[-1]\n\n        if self.bias:\n            self.b = self.add_weight((input_shape[1],),\n                                     initializer='zero',\n                                     name='{}_b'.format(self.name),\n                                     regularizer=self.b_regularizer,\n                                     constraint=self.b_constraint)\n        else:\n            self.b = None\n\n        self.built = True\n\n    def compute_mask(self, input, input_mask=None):\n        return None\n\n    def call(self, x, mask=None):\n        features_dim = self.features_dim\n        step_dim = self.step_dim\n\n        eij = K.reshape(K.dot(K.reshape(x, (-1, features_dim)),\n                        K.reshape(self.W, (features_dim, 1))), (-1, step_dim))\n\n        if self.bias:\n            eij += self.b\n\n        eij = K.tanh(eij)\n        a = K.exp(eij)\n\n        if mask is not None:\n            a *= K.cast(mask, K.floatx())\n\n        a /= K.cast(K.sum(a, axis=1, keepdims=True) + K.epsilon(), K.floatx())\n\n        a = K.expand_dims(a)\n        weighted_input = x * a\n        return K.sum(weighted_input, axis=1)\n\n    def compute_output_shape(self, input_shape):\n        return input_shape[0], self.features_dim\ndef f1(y_true, y_pred):\n\n    def recall(y_true, y_pred):\n        \n        true_positives = K.sum(K.round(K.clip(y_true*y_pred, 0, 1)))\n        possible_positives = K.sum(K.round(K.clip(y_true, 0, 1)))\n        recall = true_positives/(possible_positives + K.epsilon())\n        return recall\n\n    def precision(y_true, y_pred):\n        \n        true_positives = K.sum(K.round(K.clip(y_true*y_pred, 0, 1)))\n        predicted_positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n        precision = true_positives/(predicted_positives + K.epsilon())\n        return precision\n\n    precision = precision(y_true, y_pred)\n    recall = recall(y_true, y_pred)\n\n    return 2*((precision*recall)/(precision+recall+K.epsilon()))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Metrica F1"},{"metadata":{"trusted":true},"cell_type":"code","source":"def f1(y_true, y_pred):\n\n    def recall(y_true, y_pred):\n        \n        true_positives = K.sum(K.round(K.clip(y_true*y_pred, 0, 1)))\n        possible_positives = K.sum(K.round(K.clip(y_true, 0, 1)))\n        recall = true_positives/(possible_positives + K.epsilon())\n        return recall\n\n    def precision(y_true, y_pred):\n        \n        true_positives = K.sum(K.round(K.clip(y_true*y_pred, 0, 1)))\n        predicted_positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n        precision = true_positives/(predicted_positives + K.epsilon())\n        return precision\n\n    precision = precision(y_true, y_pred)\n    recall = recall(y_true, y_pred)\n\n    return 2*((precision*recall)/(precision+recall+K.epsilon()))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Definimos el modelo\nEl modelo a usar sera una lstm con una capa attention"},{"metadata":{"trusted":true},"cell_type":"code","source":"def model_lstm_att(embedding_matrix):\n    \n    inp = Input(shape=(maxlen,))\n    x = Embedding(max_features, embed_size, weights=[embedding_matrix], trainable=False)(inp)\n    x = Bidirectional(CuDNNLSTM(64, return_sequences=True))(x)\n    x = Bidirectional(CuDNNLSTM(32, return_sequences=True))(x)\n    \n    attn = Attention(maxlen)(x)\n    \n    y = Dense(32, activation='relu')(attn)\n    y = Dropout(0.1)(y)\n    outp = Dense(1, activation='sigmoid')(y)    \n\n    model = Model(inputs=inp, outputs=outp)\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['binary_accuracy', 'accuracy'])\n    \n    return model  ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model_lstm = model_lstm_att(embedding_matrix_glove)\nmodel_lstm.summary()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Entrenamiento"},{"metadata":{"trusted":true},"cell_type":"code","source":"def train_pred(model, epochs=2):\n    \n    for e in range(epochs):\n        model.fit(train_X, train_y, batch_size=512, epochs=3, validation_data=(val_X, val_y))\n        pred_val_y = model.predict([val_X], batch_size=1024, verbose=1)\n\n        best_thresh = 0.5\n        best_score = 0.0\n        for thresh in np.arange(0.1, 0.501, 0.01):\n            thresh = np.round(thresh, 2)\n            score = metrics.f1_score(val_y, (pred_val_y > thresh).astype(int))\n            if score > best_score:\n                best_thresh = thresh\n                best_score = score\n\n        print(\"Val puntuacion F1: {:.4f}\".format(best_score))\n\n    pred_test_y = model.predict([test_X], batch_size=1024, verbose=1)\n\n    return pred_val_y, pred_test_y, best_score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"outputs = []\npred_val_y, pred_test_y, best_score = train_pred(model_lstm, epochs=2)\noutputs.append([pred_val_y, pred_test_y, best_score, 'model_lstm_att only Glove'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"outputs.sort(key=lambda x: x[2]) \nweights = [i for i in range(1, len(outputs) + 1)]\nweights = [float(i) / sum(weights) for i in weights] \n\npred_val_y = np.mean([outputs[i][0] for i in range(len(outputs))], axis = 0)\n\nthresholds = []\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    res = metrics.f1_score(val_y, (pred_val_y > thresh).astype(int))\n    thresholds.append([thresh, res])\n    print(\"La puntuación F1 en los limites {0} y {1}\".format(thresh, res))\n    \nthresholds.sort(key=lambda x: x[1], reverse=True)\nbest_thresh = thresholds[0][0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Mejor limite:\", best_thresh, \"y puntuacion F1 \", thresholds[0][1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_test_y = np.mean([outputs[i][1] for i in range(len(outputs))], axis = 0)\npred_test_y = (pred_test_y > best_thresh).astype(int)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Revisamos el Train AUC y el Val AUC"},{"metadata":{"trusted":true},"cell_type":"code","source":"precisiones_globales=[]\ndef precision(model_lstm, registrar=False):\n    y_pred1 = model_lstm.predict(train_X)\n    train_auc1 = roc_auc_score(train_y, y_pred1)\n    y_pred1 = model_lstm.predict(val_X)\n    val_auc1 = roc_auc_score(val_y, y_pred1)\n    print('Train AUC: ', train_auc1)\n    print('Vali AUC: ', val_auc1)\n    if registrar:\n        precisiones_globales.append([train_auc1,val_auc1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"precision(model_lstm, True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Guardar el submission"},{"metadata":{"trusted":true},"cell_type":"code","source":"sub = pd.read_csv('../input/sample_submission.csv')\nout_df = pd.DataFrame({\"qid\":sub[\"qid\"].values})\nout_df['prediction'] = pred_test_y\nout_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}