{"cells":[{"metadata":{"_uuid":"c0672acd33b19b0188a94c4bf8ac1a8812b13553"},"cell_type":"markdown","source":"# Llibreries"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"from keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.initializers import he_normal, he_uniform, glorot_normal, glorot_uniform\nfrom keras import backend as K\nfrom keras.callbacks import ModelCheckpoint, ReduceLROnPlateau\nfrom keras.models import Model\nfrom keras.layers import Dense, Embedding, Bidirectional, CuDNNGRU, GlobalAveragePooling1D\nfrom keras.layers import CuDNNLSTM, GlobalMaxPooling1D, concatenate, Input, Dropout, SpatialDropout1D\nfrom keras.optimizers import Adam\nfrom keras.engine.topology import Layer\nfrom keras import initializers, regularizers, constraints, optimizers, layers\n\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nfrom sklearn.metrics import f1_score\n\nimport matplotlib.pyplot as plt\n\nfrom tqdm import tqdm\nfrom statistics import mean, median, stdev\nfrom numpy import amax, amin\n\nimport gc\nimport time\nimport pandas as pd\nimport numpy as np\nimport tensorflow as tf\nimport operator \nimport re\nimport os\nimport math","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"198ec0ca05ef5c397188832c2e9b78ffe602b9f7"},"cell_type":"markdown","source":"# Constants i Hiperparametres"},{"metadata":{"trusted":true,"_uuid":"0dc8477af3f58eaac0a458ab73ca6cd4b5744640"},"cell_type":"code","source":"batch_size = 1024\nepochs = 18\ncurrent_embd = \"Glove\"\nquestion_length =  100\nmax_eval_size = 15000\n#max_features = 50000\n\n# Cesc: Seed utilitzada pel K-Fold\nDATA_SPLIT_SEED = 2018\nK_FOLDS = 5\nK_FOLD_EPOCHS = int(epochs/K_FOLDS)\n\nseed_nb=14\nnp.random.seed(seed_nb)\ntf.set_random_seed(seed_nb)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7f321f572601ff9898d75c1831c8e4b2001cc7a3"},"cell_type":"code","source":"paraules_prohibides = ['2g1c', '2 girls 1 cup', 'acrotomophilia', 'alabama hot pocket', 'alaskan pipeline', 'anal', 'anilingus', 'anus',\n                       'apeshit', 'arsehole', 'ass', 'asshole', 'assmunch', 'auto erotic', 'autoerotic', 'babeland', 'baby batter',\n                       'baby juice', 'ball gag', 'ball gravy', 'ball kicking', 'ball licking', 'ball sack', 'ball sucking', 'bangbros',\n                       'bareback', 'barely legal', 'barenaked', 'bastard', 'bastardo', 'bastinado', 'bbw', 'bdsm', 'beaner', 'beaners',\n                       'beaver cleaver', 'beaver lips', 'bestiality', 'big black', 'big breasts', 'big knockers', 'big tits', 'bimbos',\n                       'birdlock', 'bitch', 'bitches', 'black cock', 'blonde action', 'blonde on blonde action', 'blowjob', 'blow job',\n                       'blow your load', 'blue waffle', 'blumpkin', 'bollocks', 'bondage', 'boner', 'boob', 'boobs', 'booty call',\n                       'brown showers', 'brunette action', 'bukkake', 'bulldyke', 'bullet vibe', 'bullshit', 'bung hole', 'bunghole',\n                       'busty', 'butt', 'buttcheeks', 'butthole', 'camel toe', 'camgirl', 'camslut', 'camwhore', 'carpet muncher',\n                       'carpetmuncher', 'chocolate rosebuds', 'circlejerk', 'cleveland steamer', 'clit', 'clitoris', 'clover clamps',\n                       'clusterfuck', 'cock', 'cocks', 'coprolagnia', 'coprophilia', 'cornhole', 'coon', 'coons', 'creampie', 'cum',\n                       'cumming', 'cunnilingus', 'cunt', 'darkie', 'date rape', 'daterape', 'deep throat', 'deepthroat', 'dendrophilia',\n                       'dick', 'dildo', 'dingleberry', 'dingleberries', 'dirty pillows', 'dirty sanchez', 'doggie style', 'doggiestyle',\n                       'doggy style', 'doggystyle', 'dog style', 'dolcett', 'domination', 'dominatrix', 'dommes', 'donkey punch',\n                       'double dong', 'double penetration', 'dp action', 'dry hump', 'dvda', 'eat my ass', 'ecchi', 'ejaculation',\n                       'erotic', 'erotism', 'escort', 'eunuch', 'faggot', 'fecal', 'felch', 'fellatio', 'feltch', 'female squirting',\n                       'femdom', 'figging', 'fingerbang', 'fingering', 'fisting', 'foot fetish', 'footjob', 'frotting', 'fuck',\n                       'fuck buttons', 'fuckin', 'fucking', 'fucktards', 'fudge packer', 'fudgepacker', 'futanari', 'gang bang',\n                       'gay sex', 'genitals', 'giant cock', 'girl on', 'girl on top', 'girls gone wild', 'goatcx', 'goatse', 'god damn',\n                       'gokkun', 'golden shower', 'goodpoop', 'goo girl', 'goregasm', 'grope', 'group sex', 'g-spot', 'guro', 'hand job',\n                       'handjob', 'hard core', 'hardcore', 'hentai', 'homoerotic', 'honkey', 'hooker', 'hot carl', 'hot chick', 'how to kill',\n                       'how to murder', 'huge fat', 'humping', 'incest', 'intercourse', 'jack off', 'jail bait', 'jailbait', 'jelly donut',\n                       'jerk off', 'jigaboo', 'jiggaboo', 'jiggerboo', 'jizz', 'juggs', 'kike', 'kinbaku', 'kinkster', 'kinky', 'knobbing',\n                       'leather restraint', 'leather straight jacket', 'lemon party', 'lolita', 'lovemaking', 'make me come', 'male squirting',\n                       'masturbate', 'menage a trois', 'milf', 'missionary position', 'motherfucker', 'mound of venus', 'mr hands', 'muff diver',\n                       'muffdiving', 'nambla', 'nawashi', 'negro', 'neonazi', 'nigga', 'nigger', 'nig nog', 'nimphomania', 'nipple', 'nipples',\n                       'nsfw images', 'nude', 'nudity', 'nympho', 'nymphomania', 'octopussy', 'omorashi', 'one cup two girls', 'one guy one jar',\n                       'orgasm', 'orgy', 'paedophile', 'paki', 'panties', 'panty', 'pedobear', 'pedophile', 'pegging', 'penis', 'phone sex',\n                       'piece of shit', 'pissing', 'piss pig', 'pisspig', 'playboy', 'pleasure chest', 'pole smoker', 'ponyplay', 'poof',\n                       'poon', 'poontang', 'punany', 'poop chute', 'poopchute', 'porn', 'porno', 'pornography', 'prince albert piercing',\n                       'pthc', 'pubes', 'pussy', 'queaf', 'queef', 'quim', 'raghead', 'raging boner', 'rape', 'raping', 'rapist', 'rectum',\n                       'reverse cowgirl', 'rimjob', 'rimming', 'rosy palm', 'rosy palm and her 5 sisters', 'rusty trombone', 'sadism',\n                       'santorum', 'scat', 'schlong', 'scissoring', 'semen', 'sexo', 'sexy', 'shaved beaver', 'shaved pussy',\n                       'shemale', 'shibari', 'shit', 'shitblimp', 'shitty', 'shota', 'shrimping', 'skeet', 'slanteye', 'slut', 's&m',\n                       'smut', 'snatch', 'snowballing', 'sodomize', 'sodomy', 'spic', 'splooge', 'splooge moose', 'spooge', 'spread legs',\n                       'spunk', 'strap on', 'strapon', 'strappado', 'strip club', 'style doggy', 'suck', 'sucks', 'suicide girls', 'sultry women',\n                       'swastika', 'swinger', 'tainted love', 'taste my', 'tea bagging', 'threesome', 'throating', 'tied up', 'tight white',\n                       'tit', 'tits', 'titties', 'titty', 'tongue in a', 'topless', 'tosser', 'towelhead', 'tranny', 'tribadism', 'tub girl',\n                       'tubgirl', 'tushy', 'twat', 'twink', 'twinkie', 'two girls one cup', 'undressing', 'upskirt', 'urethra play', 'urophilia',\n                       'vagina', 'venus mound', 'vibrator', 'violet wand', 'vorarephilia', 'voyeur', 'vulva', 'wank', 'wetback', 'wet dream',\n                       'white power', 'wrapping men', 'wrinkled starfish', 'xx', 'xxx', 'yaoi', 'yellow showers', 'yiffy', 'zoophilia', '🖕']\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"98c70f00a82e8aeeb88ce6b4222360343debcff7"},"cell_type":"markdown","source":"# Càrrega de dades"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")\ntest['target']=-1\ndf = pd.concat([train ,test])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6ed30af692bf5cd6e1f265bba5a9a8fa014ce793"},"cell_type":"code","source":"del train, test; gc.collect(); time.sleep(5)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"016ffc1ecda87ab11beeeb5a25a3e42cbf7ecfa4"},"cell_type":"markdown","source":"# Carregant embeddings"},{"metadata":{"trusted":true,"_uuid":"c2f939a8247a0dcb37bbdecde350e4862ae74709"},"cell_type":"code","source":"def load_embed(file):\n    def get_coefs(word,*arr): \n        return word, np.asarray(arr, dtype='float32')\n    \n    if file == '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec':\n        embeddings_index = dict(get_coefs(*o.split(\" \")) for o in tqdm(open(file)) if len(o)>100)\n    else:\n        embeddings_index = dict(get_coefs(*o.split(\" \")) for o in tqdm(open(file, encoding='latin')))\n        \n    return embeddings_index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6222c7b30f5b3a3ce9a48e285fc8f0546989e0ff"},"cell_type":"code","source":"embed_glove = load_embed('../input/embeddings/glove.840B.300d/glove.840B.300d.txt')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ce7c27140ce0acb3848a261cd138fe1790f07b8b"},"cell_type":"code","source":"embed_paragram = load_embed('../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"97ac58f93fdbfc6434d343a467911eb57c979f3e"},"cell_type":"code","source":"#embed_fasttext = load_embed('../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f3cde7414c0778e1dfc488f34d5d2d6e55c6b050"},"cell_type":"code","source":"my_embedding_matrix = dict()\n\nfor k1,v1 in embed_glove.items():\n    my_val = v1\n    if k1 in embed_paragram.keys():\n            my_val = (v1 + embed_paragram[k1])/2\n    my_embedding_matrix[k1] = my_val\n\nfor k1,v1 in embed_paragram.items():\n    if k1 not in embed_glove.keys():\n        my_embedding_matrix[k1] = v1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0e0f9bf51a642c0b2b3bc5c7ddd8e104113ffb6e"},"cell_type":"code","source":"print(len(my_embedding_matrix))\nprint(len(embed_glove))\nprint(len(embed_paragram))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7fa4a753ce22141ba91a26c38730850895938abf"},"cell_type":"code","source":"print(my_embedding_matrix[\"dog\"][0])\nprint(embed_glove[\"dog\"][0])\nprint(embed_paragram[\"dog\"][0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cfb7a8e13b0f760c912b03c2383fca187f60409d","scrolled":true},"cell_type":"code","source":"#if current_embd == \"Glove\":\n#    glove = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\n#    print(\"Extracting GloVe embedding\")\n#    embed_glove = load_embed(glove)\n#elif current_embd == \"Paragram\":\n#    paragram =  '../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt'\n#    print(\"Extracting Paragram embedding\")\n#    embed_paragram = load_embed(paragram)\n#elif current_embd == \"FastText\":\n#    wiki_news = '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'\n#    print(\"Extracting FastText embedding\")\n#    embed_fasttext = load_embed(wiki_news)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"93cfbbaf62d6952a9f31861218bdaf5939e474a3"},"cell_type":"code","source":"df['size'] = df['question_text'].str.len()\nprint(mean(df['size']))\nprint(median(df['size']))\nprint(stdev(df['size']))\nprint(amax(df['size']))\nprint(amin(df['size']))\ndf = df.drop(['size'], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aa21c74bb63d6f5aafb7544b70b5a87446b54ae2"},"cell_type":"code","source":"# Funció per crear el Vocabulari\ndef build_vocab(texts):\n    sentences = texts.apply(lambda x: x.split()).values\n    vocab = {}\n    for sentence in sentences:\n        for word in sentence:\n            try:\n                vocab[word] += 1\n            except KeyError:\n                vocab[word] = 1\n    return vocab","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0f933e142fe2e9c4912fa076aa4bef1b3d232177"},"cell_type":"code","source":"vocab = build_vocab(df['question_text'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3eab8a3780344b79f5f8dc0052b0212740313835"},"cell_type":"code","source":"print(vocab[\"Apple\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"76f58e8fc97ccef7238e5b60a2cd0714f7e2c614"},"cell_type":"code","source":"# Convertir en minuscules les paraules\ndf['question_text'] = df['question_text'].apply(lambda x: x.lower())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f83dfe17c43c28ae358121c520c592cbf83089c7"},"cell_type":"code","source":"# Afegim les paraules amb minuscules als altres diccionaris\ndef add_lower(embedding, vocab):\n    count = 0\n    for word in vocab:\n        if word in embedding and word.lower() not in embedding:  \n            embedding[word.lower()] = embedding[word]\n            count += 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2d2b627373f15a47b6a513cf9dbe580adeee79b6"},"cell_type":"code","source":"#for w in paraules_prohibides:\n#    print(w)\n\n#for q in df['question_text'].values:\n#    for w in paraules_prohibides:\n#        if w in q:\n#            df['conte_paraula_prohibida']=1\nfor index, row in df.iterrows():\n    for w in paraules_prohibides:\n        if w in row['question_text']:\n            row['conte_paraula_prohibida']=1\n            break;","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fc27db1eea1b0c323528bb5c73929fd476e75216"},"cell_type":"code","source":"if \"sexo\" in paraules_prohibides:\n    print(\"OK\")\nif \"casa\" in paraules_prohibides:\n    print(\"OK2\")    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ed88b95f3cd842831f23bee9cea7120f7808e170"},"cell_type":"code","source":"print(list(df))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d20f6f95fa766f65c05cb1a1bfc4954d8e78ecd8"},"cell_type":"code","source":"for index, row in df.iterrows():\n    if row['conte_paraula_prohibida']==1 & index < 20:\n        print (row['question_text'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6b0e06e91aca6e6a96cb39e37960237d255f20b6"},"cell_type":"code","source":"#print(my_embedding_matrix)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ef6f65dfcd365b2f8e6787b32e3aa9b9a2ff8003"},"cell_type":"code","source":"#if current_embd == \"Glove\":\n#    add_lower(embed_glove, vocab)\n#elif current_embd == \"Paragram\":\n#    add_lower(embed_paragram, vocab)\n#elif current_embd == \"FastText\":\n#    add_lower(embed_fasttext, vocab)\n\n#add_lower(embed_glove, vocab)\n#add_lower(embed_paragram, vocab)\nadd_lower(my_embedding_matrix, vocab)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"889d1639d62d48599c1f71685c3db628868814c5"},"cell_type":"code","source":"print(len(my_embedding_matrix))\nprint(len(vocab))\nprint(type(my_embedding_matrix))\nprint(type(vocab))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4d8b3bef0fcfcb07cdac70a77c9e336fadb56df4"},"cell_type":"code","source":"my_embedding_matrix = {k: v for k, v in my_embedding_matrix.items() if k in vocab}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"35d9a1b33c288d01e4880a38d4a0346269bb2036"},"cell_type":"code","source":"print(my_embedding_matrix[\"dog\"][0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4eab4ece1c779a3bce42f46a8f29445dff5ce184"},"cell_type":"code","source":"del embed_paragram, embed_glove, vocab; gc.collect(); time.sleep(5)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"249048383b5ae7ff687a39393b97892b38cd57d8"},"cell_type":"markdown","source":"# Neteja de dades"},{"metadata":{"trusted":true,"_uuid":"ac278510f63378b9d798d6921079ec1f0ddec82a"},"cell_type":"code","source":"# Contraccions\n# Cesc: He afegit canvis en monedes\n\n#contraction_mapping = {\"euros\" : \"eur\", \"dollars\": \"usd\", \"ain't\": \"is not\", \"aren't\": \"are not\",\"can't\": \"cannot\", \"'cause\": \"because\", \"could've\": \"could have\", \"couldn't\": \"could not\", \"didn't\": \"did not\",  \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\", \"he'd\": \"he would\",\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\", \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\",  \"I'd\": \"I would\", \"I'd've\": \"I would have\", \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"I've\": \"I have\", \"i'd\": \"i would\", \"i'd've\": \"i would have\", \"i'll\": \"i will\",  \"i'll've\": \"i will have\",\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'd've\": \"it would have\", \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\", \"ma'am\": \"madam\", \"mayn't\": \"may not\", \"might've\": \"might have\",\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \"must've\": \"must have\", \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \"needn't\": \"need not\", \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\", \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\", \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\", \"she'll\": \"she will\", \"she'll've\": \"she will have\", \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\", \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \"there'd've\": \"there would have\", \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\", \"they'll\": \"they will\", \"they'll've\": \"they will have\", \"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\", \"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\", \"we'll've\": \"we will have\", \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\", \"what're\": \"what are\",  \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\", \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\"you'd\": \"you would\", \"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\" }\n\n#Ho desfem perquè sembla que no xuta\ncontraction_mapping = {\"ain't\": \"is not\", \"aren't\": \"are not\",\"can't\": \"cannot\", \"'cause\": \"because\", \"could've\": \"could have\", \"couldn't\": \"could not\", \"didn't\": \"did not\",  \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\", \"he'd\": \"he would\",\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\", \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\",  \"I'd\": \"I would\", \"I'd've\": \"I would have\", \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"I've\": \"I have\", \"i'd\": \"i would\", \"i'd've\": \"i would have\", \"i'll\": \"i will\",  \"i'll've\": \"i will have\",\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'd've\": \"it would have\", \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\", \"ma'am\": \"madam\", \"mayn't\": \"may not\", \"might've\": \"might have\",\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \"must've\": \"must have\", \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \"needn't\": \"need not\", \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\", \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\", \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\", \"she'll\": \"she will\", \"she'll've\": \"she will have\", \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\", \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \"there'd've\": \"there would have\", \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\", \"they'll\": \"they will\", \"they'll've\": \"they will have\", \"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\", \"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\", \"we'll've\": \"we will have\", \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\", \"what're\": \"what are\",  \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\", \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\"you'd\": \"you would\", \"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\" }","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4635dbfc3444e89ac4a510055c80992bd2abe1a3"},"cell_type":"code","source":"# Netejar contraccions\ndef clean_contractions(text, mapping):\n    specials = [\"’\", \"‘\", \"´\", \"`\"]\n    for s in specials:\n        text = text.replace(s, \"'\")\n    text = ' '.join([mapping[t] if t in mapping else t for t in text.split(\" \")])\n    return text","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3a0a00286fd6b94b978a25580942ff2cc8811c8a"},"cell_type":"code","source":"df['question_text'] = df['question_text'].apply(lambda x: clean_contractions(x, contraction_mapping))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3edf591ceb3a145a2bcb4befe94d9e92bff8d096"},"cell_type":"code","source":"# Caracters especials de puntuació\npunct = \"/-'?!.,#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~\" + '\"\"“”’' + '∞θ÷α•à−β∅³π‘₹´°£€\\×™√²—–&'\n\n# Cesc: Afegim uns quants caracters mes que he trobat#\n#mes_punct = \"[,.:)(-!?|;$&/[]>%=#*+\\\\•~@£·_{}©^®`<→°€™›♥←×§″′Â█½à…“★”–●â►−¢²¬░¶↑±¿▾═¦║―¥▓—‹─▒：¼⊕▼▪†■’▀¨▄♫☆é¯♦¤▲è¸¾Ã⋅‘∞∙）↓、│（»，♪╩╚³・╦╣╔╗▬❤ïØ¹≤‡√])\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5da633d5f3367c676dd0c960498c8d43c5d47884"},"cell_type":"code","source":"df['size'] = df['question_text'].str.len()\nprint(mean(df['size']))\nprint(median(df['size']))\nprint(stdev(df['size']))\nprint(amax(df['size']))\nprint(amin(df['size']))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3e85e7f031561f1b702d44aa356290978f0a1b93"},"cell_type":"code","source":"# Reemplacament dels caracters especials\n# Cesc: He afegit canvis en monedes\npunct_mapping = {\"‘\": \"'\", \"₹\": \"e\", \"´\": \"'\", \"°\": \"\", \"€\": \"eur\",\"$\": \"usd\",  \"™\": \"tm\", \"√\": \" sqrt \", \"×\": \"x\", \"²\": \"2\", \"—\": \"-\", \"–\": \"-\", \"’\": \"'\", \"_\": \"-\", \"`\": \"'\", '“': '\"', '”': '\"', '“': '\"', \"£\": \"e\", '∞': 'infinity', 'θ': 'theta', '÷': '/', 'α': 'alpha', '•': '.', 'à': 'a', '−': '-', 'β': 'beta', '∅': '', '³': '3', 'π': 'pi', }\n#punct_mapping = {\"‘\": \"'\", \"₹\": \"e\", \"´\": \"'\", \"°\": \"\", \"€\": \"e\", \"™\": \"tm\", \"√\": \" sqrt \", \"×\": \"x\", \"²\": \"2\", \"—\": \"-\", \"–\": \"-\", \"’\": \"'\", \"_\": \"-\", \"`\": \"'\", '“': '\"', '”': '\"', '“': '\"', \"£\": \"e\", '∞': 'infinity', 'θ': 'theta', '÷': '/', 'α': 'alpha', '•': '.', 'à': 'a', '−': '-', 'β': 'beta', '∅': '', '³': '3', 'π': 'pi', }","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"705f82412fbbfdbfdc15eaee5ee3a93470a4add3"},"cell_type":"code","source":"# Neteja dels caracters especials\ndef clean_special_chars(text, punct, mapping):\n    for p in mapping:\n        text = text.replace(p, mapping[p])\n\n    for p in punct:\n        text = text.replace(p, f' {p} ')\n        #text = text.replace(p, '')\n    specials = {'\\u200b': ' ', '…': ' ... ', '\\ufeff': '', 'करना': '', 'है': ''}  # Other special characters that I have to deal with in last\n    for s in specials:\n        text = text.replace(s, specials[s])\n    \n    return text","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3f14c701e528b96ef5daca8847959cd01654b3b2"},"cell_type":"code","source":"df['question_text'] = df['question_text'].apply(lambda x: clean_special_chars(x, punct, punct_mapping))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"62a92431279f0865c7b3a26a816073ada0f2ddb5"},"cell_type":"code","source":"# Cesc: Ho apliquem també amb la nova llista de caracters raros\n#df['question_text'] = df['question_text'].apply(lambda x: clean_special_chars(x, mes_punct, punct_mapping))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2b8c795b29dec685cadabad3980fec37aecd14f5"},"cell_type":"code","source":"# Reemplaçament d'errors ortogràfics\nmispell_dict = {'colour': 'color', 'centre': 'center', 'favourite': 'favorite', 'travelling': 'traveling', 'counselling': 'counseling', 'theatre': 'theater', 'cancelled': 'canceled', 'labour': 'labor', 'organisation': 'organization', 'wwii': 'world war 2', 'citicise': 'criticize', 'youtu ': 'youtube ', 'Qoura': 'Quora', 'sallary': 'salary', 'Whta': 'What', 'narcisist': 'narcissist', 'howdo': 'how do', 'whatare': 'what are', 'howcan': 'how can', 'howmuch': 'how much', 'howmany': 'how many', 'whydo': 'why do', 'doI': 'do I', 'theBest': 'the best', 'howdoes': 'how does', 'mastrubation': 'masturbation', 'mastrubate': 'masturbate', \"mastrubating\": 'masturbating', 'pennis': 'penis', 'Etherium': 'Ethereum', 'narcissit': 'narcissist', 'bigdata': 'big data', '2k17': '2017', '2k18': '2018', 'qouta': 'quota', 'exboyfriend': 'ex boyfriend', 'airhostess': 'air hostess', \"whst\": 'what', 'watsapp': 'whatsapp', 'demonitisation': 'demonetization', 'demonitization': 'demonetization', 'demonetisation': 'demonetization'}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bba7f5f2c10ac2366c2b9dab9915d45ca236ad8b"},"cell_type":"code","source":"# Correcció d'errors ortogràfics\ndef correct_spelling(x, dic):\n    for word in dic.keys():\n        x = x.replace(word, dic[word])\n    return x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"deeee42835b50577beaf7f028d227290d65e6253"},"cell_type":"code","source":"df['question_text'] = df['question_text'].apply(lambda x: correct_spelling(x, mispell_dict))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"be5908a06be5bbd64e531c36762d020b037a6ab6"},"cell_type":"code","source":"df['size'] = df['question_text'].str.len()\nprint(mean(df['size']))\nprint(median(df['size']))\nprint(stdev(df['size']))\nprint(amax(df['size']))\nprint(amin(df['size']))\ndf = df.drop(['size'],axis=1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cef2e5c37b01a92df3ccad6795c3b6c15c4086f0"},"cell_type":"markdown","source":"# Tokenització"},{"metadata":{"trusted":true,"_uuid":"f7a5ecfea1cf16ed84d8443f5059ffb1e94e92c9"},"cell_type":"code","source":"total_set_qid = df['qid'].values\ntotal_set_X = df[\"question_text\"].fillna(\"_na_\").values\ntotal_set_y = df['target'].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3d0eb13fa6e90eb7210ea678582b9d8090c7caf7"},"cell_type":"code","source":"## Tokenize the sentences\ntokenizer = Tokenizer(num_words=None, filters='')\ntokenizer.fit_on_texts(list(total_set_X))\ntotal_set_X = tokenizer.texts_to_sequences(total_set_X)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a18855b94a28804a2a122743a64975d09b877847"},"cell_type":"code","source":"## Pad the sentences \ntotal_set_X = pad_sequences(total_set_X, maxlen=question_length)\n# TODO: provar amb opcio padding='post' per posar els zeros a la dreta","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5153b69cb9be7ca91689dd9d53e21a8ad3e99cb6"},"cell_type":"code","source":"df = pd.concat([pd.DataFrame(total_set_qid) ,pd.DataFrame(total_set_X), pd.DataFrame(total_set_y)], axis=1, keys=[\"qid\", \"question_text\", \"target\"])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7b688bb7634f28731cbc39399a460b066c22acfc"},"cell_type":"markdown","source":"# Creació de la matriu d' embeddings"},{"metadata":{"trusted":true,"_uuid":"4729e5e9393a10f04da966ce7c53700e2375386a"},"cell_type":"code","source":"def embedding_matrix_creator(embeddings_index):\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n\n    word_index = tokenizer.word_index\n    nb_words = len(word_index) #min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words+1, embed_size))\n    for word, i in word_index.items():\n        #if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n    return embedding_matrix","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aa99c7dfdd7d61d77baf48107a3c526f4feee010"},"cell_type":"code","source":"print(len(my_embedding_matrix))\n#print(my_embedding_matrix[\"apple\"])\nprint(type(my_embedding_matrix))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8e02e8127f020584b41ffcc5f3f0726a7ccf651e"},"cell_type":"code","source":"#emb_matrix_fasttext = embedding_matrix_creator(embed_fasttext)\n#emb_matrix_glove = embedding_matrix_creator(embed_glove)\n#print(len(emb_matrix_fasttext))\n#print(len(emb_matrix_glove))\nemb_matrix = embedding_matrix_creator(my_embedding_matrix)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"425c6bf0d5b9690ba94422d68c4c27970e314943"},"cell_type":"code","source":"#if current_embd == \"Glove\":\n#    emb_matrix = emdedding_matrix_creator(embed_glove)\n#    del embed_glove\n#elif current_embd == \"Paragram\":\n#    emb_matrix = emdedding_matrix_creator(embed_paragram)\n#    del embed_paragram\n#elif current_embd == \"FastText\":\n#    emb_matrix = emdedding_matrix_creator(embed_fasttext)\n#    del embed_fasttext\n\n# Fem la mitjana de tots els 3 embeddings\n#emb_matrix = np.mean([ emdedding_matrix_creator(embed_glove),  \n#                       emdedding_matrix_creator(embed_paragram),\n#                       emdedding_matrix_creator(embed_fasttext)], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fd1b59757ca5b17df8a3b87cbe1107ad3a071783"},"cell_type":"code","source":"#del embed_glove; del embed_paragram; del embed_fasttext; gc.collect(); time.sleep(10)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"45a02bbe379743e6fc208dc8f4c73dfefe2be90e"},"cell_type":"markdown","source":"# Separació de Sets"},{"metadata":{"trusted":true,"_uuid":"0b00c41a51cc1716094999f430e4b0e78892abeb"},"cell_type":"code","source":"# Creem el train set i el eval set\ntrain_df, val_df = train_test_split(df[df.target[0]!=-1], test_size=0.1)\n\ntrain_X = np.array(train_df[\"question_text\"])\ntrain_y = np.array(train_df[\"target\"])\n\nval_X = np.array(val_df[\"question_text\"])\nval_y = np.array(val_df[\"target\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a38dfe1d5c74255bcec147ee4c5f757437f6d39e"},"cell_type":"code","source":"# Creem el test set\ntest_df = df[df.target[0]==-1]\n#test_df = test_df.drop('target', axis=1)\n\ntest_X=np.array(test_df[\"question_text\"])\n#print(test_X.shape)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"343b555b4d7d7e79e5c55744a0329e084849b42d"},"cell_type":"markdown","source":"# Training"},{"metadata":{"_uuid":"a811edaefd41dfd77651b33440bf826920c4ba79"},"cell_type":"markdown","source":"## F1 Score"},{"metadata":{"trusted":true,"_uuid":"f74dc586581dadb38c78345750e3e046a87ff93b"},"cell_type":"code","source":"def f1(y_true, y_pred):\n    def recall(y_true, y_pred):\n        true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n        possible_positives = K.sum(K.round(K.clip(y_true, 0, 1)))\n        recall = true_positives / (possible_positives + K.epsilon())\n        return recall\n\n    def precision(y_true, y_pred):\n        true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n        predicted_positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n        precision = true_positives / (predicted_positives + K.epsilon())\n        return precision\n    \n    precision = precision(y_true, y_pred)\n    recall = recall(y_true, y_pred)\n    return 2*((precision*recall)/(precision+recall+K.epsilon()))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"29b4daa45cf849a5b0d8f760a78eab5f4331835e"},"cell_type":"markdown","source":"# Custom Attention Layer"},{"metadata":{"trusted":true,"_uuid":"c605669712bae344867322d9744b6e9a59ff56e5"},"cell_type":"code","source":"class Attention(Layer):\n    def __init__(self, step_dim,\n                 W_regularizer=None, b_regularizer=None,\n                 W_constraint=None, b_constraint=None,\n                 bias=True, **kwargs):\n        self.supports_masking = True\n        self.init = initializers.get('glorot_uniform')\n\n        self.W_regularizer = regularizers.get(W_regularizer)\n        self.b_regularizer = regularizers.get(b_regularizer)\n\n        self.W_constraint = constraints.get(W_constraint)\n        self.b_constraint = constraints.get(b_constraint)\n\n        self.bias = bias\n        self.step_dim = step_dim\n        self.features_dim = 0\n        super(Attention, self).__init__(**kwargs)\n\n    def build(self, input_shape):\n        assert len(input_shape) == 3\n\n        self.W = self.add_weight((input_shape[-1],),\n                                 initializer=self.init,\n                                 name='{}_W'.format(self.name),\n                                 regularizer=self.W_regularizer,\n                                 constraint=self.W_constraint)\n        self.features_dim = input_shape[-1]\n\n        if self.bias:\n            self.b = self.add_weight((input_shape[1],),\n                                     initializer='zero',\n                                     name='{}_b'.format(self.name),\n                                     regularizer=self.b_regularizer,\n                                     constraint=self.b_constraint)\n        else:\n            self.b = None\n\n        self.built = True\n\n    def compute_mask(self, input, input_mask=None):\n        return None\n    \n    def call(self, x, mask=None):\n        features_dim = self.features_dim\n        step_dim = self.step_dim\n\n        eij = K.reshape(K.dot(K.reshape(x, (-1, features_dim)),\n                        K.reshape(self.W, (features_dim, 1))), (-1, step_dim))\n\n        if self.bias:\n            eij += self.b\n\n        eij = K.tanh(eij)\n\n        a = K.exp(eij)\n\n        if mask is not None:\n            a *= K.cast(mask, K.floatx())\n\n        a /= K.cast(K.sum(a, axis=1, keepdims=True) + K.epsilon(), K.floatx())\n\n        a = K.expand_dims(a)\n        weighted_input = x * a\n        return K.sum(weighted_input, axis=1)\n\n    def compute_output_shape(self, input_shape):\n        return input_shape[0],  self.features_dim","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d7a2db7dfd3ccf9a844dd0d8a919233255bdd3e8"},"cell_type":"markdown","source":"# Model 1"},{"metadata":{"trusted":true,"_uuid":"4ced204f9460bc1adeedb1ecddce7692361ff852"},"cell_type":"code","source":"def make_old_model(embedding_matrix, embed_size=300, loss='binary_crossentropy'):\n    inp    = Input(shape=(question_length,))\n    x      = Embedding(embedding_matrix.shape[0], embed_size, weights=[embedding_matrix], trainable=False)(inp)\n    x      = Bidirectional(CuDNNGRU(128, return_sequences=True))(x)\n    x      = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\n    avg_pl = GlobalAveragePooling1D()(x)\n    max_pl = GlobalMaxPooling1D()(x)\n    concat = concatenate([avg_pl, max_pl])\n    dense  = Dense(64, activation=\"relu\")(concat)\n    drop   = Dropout(0.1)(concat)\n    output = Dense(1, activation=\"sigmoid\")(concat)\n    \n    model  = Model(inputs=inp, outputs=output)\n    model.compile(loss=loss, optimizer=Adam(lr=0.0001), metrics=['accuracy', f1])\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ba1592a0853927696b06d0ba7ed5aed305a60583"},"cell_type":"code","source":"#model = make_model(emb_matrix)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"743a39adb617fb17dd4781e4d88f68762852e975"},"cell_type":"markdown","source":"# Model 2"},{"metadata":{"trusted":true,"_uuid":"3a703bfd6983e3fb2065a6ceadc6d4a156217cf6"},"cell_type":"code","source":"def model_lstm_gru_atten(embedding_matrix, embed_size=300, loss='binary_crossentropy'):\n    inp = Input(shape=(question_length,))\n    x = Embedding(embedding_matrix.shape[0], embed_size, weights=[embedding_matrix], trainable=False)(inp)\n    x = SpatialDropout1D(0.1,seed=seed_nb)(x)\n    x = Bidirectional(CuDNNLSTM(64, kernel_initializer=glorot_uniform(seed=seed_nb), return_sequences=True))(x)\n    y = Bidirectional(CuDNNGRU(40,kernel_initializer=glorot_uniform(seed=seed_nb), return_sequences=True))(x)\n\n    atten_1 = Attention(question_length)(x) \n    atten_2 = Attention(question_length)(y)\n    avg_pool = GlobalAveragePooling1D()(y)\n    max_pool = GlobalMaxPooling1D()(y)\n\n    conc = concatenate([atten_1, atten_2, avg_pool, max_pool])\n    conc = Dense(16,kernel_initializer=he_uniform(seed=seed_nb),  activation=\"relu\")(conc)\n    conc = Dropout(0.1,seed=seed_nb)(conc)\n    outp = Dense(1,kernel_initializer=he_uniform(seed=seed_nb),  activation=\"sigmoid\")(conc)    \n\n    model = Model(inputs=inp, outputs=outp)\n    model.compile(loss=loss, optimizer=Adam(lr=0.0001), metrics=['accuracy', f1])\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"42c220ac91f1bb66056f50c2f3324537f7359212"},"cell_type":"code","source":"#model = model_lstm_gru_atten(emb_matrix)\nmodel = make_old_model(emb_matrix)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"041a2b9f49ae3d584078a155a78881ce9feda8e8"},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"84dd871f382ed34cf805121134972291758c0dd0"},"cell_type":"markdown","source":"# Callbacks"},{"metadata":{"trusted":true,"_uuid":"15303ea49d116457f16f92202806ca690ebf775c"},"cell_type":"code","source":"checkpoints = ModelCheckpoint('weights.hdf5', monitor=\"val_f1\", mode=\"max\", verbose=True, save_best_only=True)\nreduce_lr = ReduceLROnPlateau(monitor='val_f1', factor=0.1, patience=2, verbose=1, min_lr=0.000001)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"92c04cf93171d94e5a9cabd4a394599bdfe7acaf"},"cell_type":"markdown","source":"# Training, Eval Prediction & Test Prediction"},{"metadata":{"trusted":true,"_uuid":"c39fecfe792be457e0d53dca64320952930f0bdc"},"cell_type":"code","source":"# Trobar el threshold més óptim\ndef tweak_threshold(pred, truth):\n    thresholds = []\n    scores = []\n    print(\"Threshold: Valor\")\n    for thresh in np.arange(0.01, 1.01, 0.01):\n        thresh = np.round(thresh, 2)\n        thresholds.append(thresh)\n        score = f1_score(truth, (pred>thresh).astype(int))\n        print(thresh, \": \", score)\n        scores.append(score)\n    return np.max(scores), thresholds[np.argmax(scores)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3093a6b87423533bd2866067162f1b440be6ccff"},"cell_type":"code","source":"#print(history.history)\n#plt.plot(history.history['acc'])\n#plt.plot(history.history['f1'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"8c843883f97716df8367df5d314abf8e75a4bb9f"},"cell_type":"code","source":"df_X =  df[df.target[0]!=-1][\"question_text\"]\ndf_y =  df[df.target[0]!=-1][\"target\"]\n\n# Cesc: Afegim la separacio de K-Folds\ntrain_meta = np.zeros(df_y.shape)\ntest_meta = np.zeros(test_X.shape[0])\n\n#K_FOLDS = 4\n#K_FOLD_EPOCHS = 1 #int(epochs/K_FOLDS)\nmy_splits = list(StratifiedKFold(n_splits=K_FOLDS,\n                                  shuffle=True,\n                                  random_state=DATA_SPLIT_SEED).split(df_X, df_y))\n\nfor idx, (train_idx, valid_idx) in enumerate(my_splits):\n    print(\"======== K-FOLD: {0} ==========\".format(idx))\n    train_X = df_X.iloc[train_idx]\n    train_y = df_y.iloc[train_idx]\n    val_X = df_X.iloc[valid_idx]\n    val_y = df_y.iloc[valid_idx]\n#    model = model_lstm_gru_atten(emb_matrix)\n    model = make_old_model(emb_matrix)\n#   pred_val_y, pred_test_y, best_score = train_pred(model, X_train, y_train, X_val, y_val, epochs = 8, callback = [clr,])\n    history = model.fit(train_X, train_y, batch_size=batch_size, epochs=K_FOLD_EPOCHS, validation_data=[val_X, val_y], callbacks=[checkpoints, reduce_lr])\n    model.load_weights('weights.hdf5')\n    eval_pred = model.predict(val_X, batch_size=batch_size, verbose=1)\n    test_pred = model.predict(test_X, batch_size=batch_size, verbose=1)\n   # pred_val_y = model.predict([val_X], batch_size=batch_size, verbose=0)\n    train_meta[valid_idx] = eval_pred#.reshape(-1)\n    test_meta += test_pred.reshape(-1) / len(my_splits)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"56515092a6646ac05196750447e99cf5d1e652a1"},"cell_type":"code","source":"score_val, best_thresh = tweak_threshold(train_meta, df_y)\nprint(\"=====================================\")\nprint(f\"Scored {round(score_val, 4)} for threshold {best_thresh} with untreated texts on validation data\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9dd0665d0a407e94ef702aa24d12b7385c6093ec"},"cell_type":"markdown","source":"# Output"},{"metadata":{"trusted":true,"_uuid":"5eb936d7d5a0415a60f6b4cc3008d230d97a2008"},"cell_type":"code","source":"# Imprimim la submission en un fitxer\ny_te = (np.array(test_meta) > best_thresh).astype(np.int)\nqid = test_df[\"qid\"].values\n#submit_df = pd.DataFrame({\"qid\": test_df[\"qid\"], \"prediction\": y_te})\n#submit_df = pd.concat([pd.DataFrame(qid),pd.DataFrame(y_te)], axis = 1, keys=[\"qid\", \"prediction\"])\nsubmit_df = pd.concat([pd.DataFrame(qid, columns=['qid']),pd.DataFrame(y_te, columns=['prediction'])], axis = 1)\nsubmit_df.to_csv(\"submission_cesc.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c0fe8ba963a2db1569a49ab8a4efe8ac80e6fd79"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}