{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nimport time\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nfrom keras.engine.topology import Layer\nimport math\nimport operator \nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D, TimeDistributed, CuDNNLSTM,Conv2D, SpatialDropout1D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D, GlobalAveragePooling1D, concatenate, Flatten, Reshape, AveragePooling2D, Average, BatchNormalization\nfrom keras.models import Model\n#from keras.layers import Wrapper\nimport keras.backend as K\nfrom keras.optimizers import Adam\nfrom keras import initializers, regularizers, constraints, optimizers, layers\nimport re\nimport gc\nfrom sklearn.preprocessing import StandardScaler\ntqdm.pandas()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":" \nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\n\nimport os\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport math\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nimport re\nfrom nltk.corpus import stopwords\nstop = set(stopwords.words('english'))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport math\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Obtener los datos"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(\"../input/quora-insincere-questions-classification/train.csv\")\ntest_df = pd.read_csv('../input/quora-insincere-questions-classification/test.csv')\n\nprint(\"Train shape: \",train_df.shape)\nprint(\"Test shape: \",test_df.shape)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Preprocesamiento de la data"},{"metadata":{},"cell_type":"markdown","source":"Diccionarios para hacer limpieza de texto"},{"metadata":{"trusted":true},"cell_type":"code","source":"contraction_mapping = {\"ain't\": \"is not\", \"aren't\": \"are not\",\"can't\": \"cannot\", \"'cause\": \"because\", \"could've\": \"could have\", \"couldn't\": \"could not\",\n                       \"didn't\": \"did not\",  \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\",\n                       \"he'd\": \"he would\",\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\", \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\",\n                       \"I'd\": \"I would\", \"I'd've\": \"I would have\", \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"I've\": \"I have\", \"i'd\": \"i would\",\n                       \"i'd've\": \"i would have\", \"i'll\": \"i will\",  \"i'll've\": \"i will have\",\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\",\n                       \"it'd've\": \"it would have\", \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\", \"ma'am\": \"madam\", \"mayn't\": \"may not\",\n                       \"might've\": \"might have\",\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \"must've\": \"must have\", \"mustn't\": \"must not\", \"mustn't've\": \"must not have\",\n                       \"needn't\": \"need not\", \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\", \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\",\n                       \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\", \"she'll\": \"she will\", \"she'll've\": \"she will have\",\n                       \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\", \"so've\": \"so have\",\"so's\": \"so as\",\n                       \"this's\": \"this is\",\"that'd\": \"that would\", \"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \"there'd've\": \"there would have\",\n                       \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\", \"they'll\": \"they will\", \"they'll've\": \"they will have\",\n                       \"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\", \"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\",\n                       \"we'll've\": \"we will have\", \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\",\n                       \"what're\": \"what are\",  \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\",\n                       \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\",\n                       \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\", \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\",\n                       \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\"you'd\": \"you would\",\n                       \"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\"}\n\ncontraction_patterns = [ (r'won\\'t', 'will not'), (r'can\\'t', 'cannot'), (r'i\\'m', 'i am'), (r'ain\\'t', 'is not'), (r'(\\w+)\\'ll', '\\g<1> will'), (r'(\\w+)n\\'t', '\\g<1> not'),\n                         (r'(\\w+)\\'ve', '\\g<1> have'), (r'(\\w+)\\'s', '\\g<1> is'), (r'(\\w+)\\'re', '\\g<1> are'), (r'(\\w+)\\'d', '\\g<1> would'), (r'&', 'and'), (r'dammit', 'damn it'),\n                        (r'dont', 'do not'), (r'wont', 'will not') ]\n\npunct_mapping = {\"‘\": \"'\", \"₹\": \"e\", \"´\": \"'\", \"°\": \"\", \"€\": \"e\", \"™\": \"tm\", \"√\": \" sqrt \", \"×\": \"x\", \"²\": \"2\", \"—\": \"-\", \"–\": \"-\", \"’\": \"'\", \"_\": \"-\", \"`\": \"'\", '“': '\"', '”': '\"', '“': '\"', \"£\": \"e\", '∞': 'infinity', 'θ': 'theta', '÷': '/', 'α': 'alpha', '•': '.', 'à': 'a', '−': '-', 'β': 'beta', '∅': '', '³': '3', 'π': 'pi', }\npunct = \"/-'?!.,#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~\" + '\"\"“”’' + '∞θ÷α•à−β∅³π‘₹´°£€\\×™√²—–&'\n\nmispell_dict = {'advanatges': 'advantages', 'irrationaol': 'irrational' , 'defferences': 'differences','lamboghini':'lamborghini','hypothical':'hypothetical', 'colour': 'color',\n                'centre': 'center', 'favourite': 'favorite', 'travelling': 'traveling', 'counselling': 'counseling', 'theatre': 'theater', 'cancelled': 'canceled', 'labour': 'labor',\n                'organisation': 'organization', 'wwii': 'world war 2', 'citicise': 'criticize', 'youtu ': 'youtube ', 'Qoura': 'Quora', 'qoura' : 'quora', 'sallary': 'salary', 'Whta': 'What',\n                'narcisist': 'narcissist', 'howdo': 'how do', 'whatare': 'what are', 'howcan': 'how can', 'howmuch': 'how much', 'howmany': 'how many', 'whydo': 'why do',\n                'doI': 'do I', 'theBest': 'the best', 'howdoes': 'how does', 'mastrubation': 'masturbation', 'mastrubate': 'masturbate', \"mastrubating\": 'masturbating',\n                'pennis': 'penis', 'Etherium': 'Ethereum', 'etherium': 'ethereum', 'narcissit': 'narcissist', 'bigdata': 'big data', '2k17': '2017', '2k18': '2018', 'qouta': 'quota',\n                'exboyfriend': 'ex boyfriend', 'airhostess': 'air hostess', \"whst\": 'what', 'watsapp': 'whatsapp', 'demonitisation': 'demonetization',\n                'demonitization': 'demonetization', 'demonetisation': 'demonetization', 'pokémon': 'pokemon'}","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Métodos"},{"metadata":{"trusted":true},"cell_type":"code","source":"#metodo para sustituir las diferentes tildes por '\ndef clean_contractions(text, mapping):\n    specials = [\"’\", \"‘\", \"´\", \"`\"]\n    for s in specials:\n        text = text.replace(s, \"'\")\n        #si t esta en mapping agregar la nueva sustitucion\n        #si no dejarla igual\n    text = ' '.join([mapping[t] if t in mapping else t for t in text.split(\" \")])\n    return text\n#replace the constractions in text\ndef replaceContraction(text):\n    patterns = [(re.compile(regex), repl) for (regex, repl) in contraction_patterns]\n    for (pattern, repl) in patterns:\n        (text, count) = re.subn(pattern, repl, text)\n    return text\n\n#clean text\n#reemplazando caracteres \ndef clean_text(x):\n    x = str(x)\n    #se reemplazan estos caracteres por un espacio\n    for punct in \"/-'\":\n        x = x.replace(punct, ' ')\n    #si un & esta pegado a una palabra se agrega un espacio\n    for punct in '&':\n        x = x.replace(punct, f' {punct} ')\n    #cualquiera de estos simbolos que se encuentre en la data\n    #sera reemplazado por ''\n    for punct in '?!.,\"#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~' + '“”’':\n        x = x.replace(punct, '')\n    return x\n\n#numeros mayores a 2 digitos se les coloca e nombre de 'number'\ndef clean_numbers(x):\n    x = re.sub('[0-9]{5,}', ' number ', x)\n    x = re.sub('[0-9]{4}', ' number ', x)\n    x = re.sub('[0-9]{3}', ' number ', x)\n    x = re.sub('[0-9]{2}', ' number ', x)\n    return x\n\n\n\n#replace special characters\ndef clean_special_chars(text, punct, mapping):\n    for p in mapping:\n        text = text.replace(p, mapping[p])\n    for p in punct:\n        text = text.replace(p, f' {p} ')\n    specials = {'\\u200b': ' ', '…': ' ... ', '\\ufeff': '', 'करना': '', 'है': ''}  # Other special characters\n    for s in specials:\n        text = text.replace(s, specials[s])\n    return text\n\n\n#por cada palabra mal escrita se reemplaza por la palabra bien escrita\n#ese dictionario esta en misspell_dict\ndef correct_spelling(x, dic):\n    for word in dic.keys():\n        x = x.replace(word, dic[word])\n    return x","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Embedding"},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_embed(file):\n    def get_coefs(word,*arr): \n        return word, np.asarray(arr, dtype='float32')\n    \n    if file == '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec':\n        embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(file) if len(o)>100)\n    else:\n        embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(file, encoding='latin'))\n        \n    return embeddings_index","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Se recibe y se crea embedding"},{"metadata":{"trusted":true},"cell_type":"code","source":"glove = '../input/quora-insincere-questions-classification/embeddings/glove.840B.300d/glove.840B.300d.txt'\nembed_glove = load_embed(glove)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(embed_glove)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Aplicando funciones de limpieza a traindf y testdf por separado"},{"metadata":{"trusted":true},"cell_type":"code","source":"# poner en minusculas\ntrain_df['treated_question'] = train_df['question_text'].progress_apply(lambda x: x.lower())\n# aplicar contracciones\ntrain_df['treated_question'] = train_df['treated_question'].progress_apply(lambda x: clean_contractions(x, contraction_mapping))\n# special characters\ntrain_df['treated_question'] = train_df['treated_question'].progress_apply(lambda x: clean_special_chars(x, punct, punct_mapping))\n# aplicar funcion correct_spelling\ntrain_df['treated_question'] = train_df['treated_question'].progress_apply(lambda x: correct_spelling(x, mispell_dict))\n#aplicar funcion clean_numbers\ntrain_df['treated_question'] = train_df['treated_question'].apply(lambda x: clean_numbers(x))\n\n\n# poner en minuscula\ntest_df['treated_question'] = test_df['question_text'].progress_apply(lambda x: x.lower())\n# Contracciones\ntest_df['treated_question'] = test_df['treated_question'].progress_apply(lambda x: clean_contractions(x, contraction_mapping))\n# special characters\ntest_df['treated_question'] = test_df['treated_question'].progress_apply(lambda x: clean_special_chars(x, punct, punct_mapping))\n# aplicar funcion correct_spelling\ntest_df['treated_question'] = test_df['treated_question'].progress_apply(lambda x: correct_spelling(x, mispell_dict))\ntest_df['treated_question'] = test_df['treated_question'].apply(lambda x: clean_numbers(x))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Dividir train_df en entrenamiento y validacion"},{"metadata":{"trusted":true},"cell_type":"code","source":"train, val = train_test_split(train_df, test_size=0.2, random_state=3)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Rellenar valores nulos"},{"metadata":{"trusted":true},"cell_type":"code","source":"xtrain = train['question_text'].fillna('_na_').values\nxval = val['question_text'].fillna('_na_').values\nxtest = test_df['question_text'].fillna('_na_').values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(xtrain)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#numero de dimensiones del embedding\nembed_size = 300\n#numero de palabras en diccionario\nmax_features = 10000\n#maximo numero de palabras que se analizaran\nmaxlen = 80\n\n\n#e diccionario va a tener 10.000 palabras\n#se tendran 10.000 guardadas basado en la frecuencia en que ocurren\n#solamente se guardaran las 10.000 palabras mas frecuentes\ntokenizer = Tokenizer(num_words=max_features)\n\ntokenizer.fit_on_texts(list(xtrain))\n\nxtrain = tokenizer.texts_to_sequences(xtrain)\nxval = tokenizer.texts_to_sequences(xval)\nxtest = tokenizer.texts_to_sequences(xtest)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":" xtrain, xval, xtest por cada pregunta tienen numeros que representan el indice de las palabras"},{"metadata":{},"cell_type":"markdown","source":"Hacer padding, se hace padding para que xtrain, xval y xtest tenga como maximo 80 palabras"},{"metadata":{"trusted":true},"cell_type":"code","source":"\nxtrain = pad_sequences(xtrain, maxlen=maxlen)\nxval = pad_sequences(xval,maxlen=maxlen)\nxtest = pad_sequences(xtest,maxlen=maxlen)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ytrain = train['target'].values\nyval = val['target'].values","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Ahora la matriz de embeddings"},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_glove_matrix(word_index, embeddings_index):\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean, emb_std = all_embs.mean(), all_embs.std()\n    EMBED_SIZE = all_embs.shape[1]\n    \n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n\n    for word, i in word_index.items():\n        if i >= max_features:\n            continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None:\n            embedding_matrix[i] = embedding_vector\n\n    return embedding_matrix","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"np.random.seed(2)\n\ntrn_idx = np.random.permutation(len(xtrain))\nval_idx = np.random.permutation(len(xval))\n\nxtrain = xtrain[trn_idx]\nytrain = ytrain[trn_idx]\nxval = xval[val_idx]\nyval = yval[val_idx]\n\nembedding_matrix_glove = load_glove_matrix(tokenizer.word_index, embed_glove)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def model():\n    inp = Input(shape=(maxlen,))\n    x = Embedding(max_features, embed_size, weights=[embedding_matrix_glove], trainable=True)(inp)\n    x = SpatialDropout1D(0.15)(x)\n    x = Bidirectional(LSTM(128, return_sequences=True))(x)\n    x = Conv1D(filters=64, kernel_size=1)(x)\n    x = GlobalMaxPool1D()(x)\n    x_f = Dense(128, activation=\"relu\")(x)\n    x_f = Dropout(0.15)(x_f)\n    x_f = BatchNormalization()(x_f)\n    x_f = Dense(1, activation=\"sigmoid\")(x_f)\n    \n    model = Model(inputs=inp, outputs = x_f)\n    model.compile(loss='binary_crossentropy', optimizer=Adam(lr=0.0010), metrics=['binary_accuracy'])\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model1 = model()\nprint(model1.summary())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model1.fit(xtrain,ytrain, batch_size = 512, epochs= 2, validation_data=(xval,yval))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"prediction = model1.predict(xtest, batch_size=1024)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"out_df = pd.DataFrame({\"qid\":test_df[\"qid\"].values})\nout_df['prediction'] = prediction\nout_df.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}