rm(list=ls())

library(keras)
library(textclean)
library(tidyverse)
library(qdapRegex)
library(data.table)
library(tensorflow)
library(stringi)
library(stringr)
library(tidytext)
library(stopwords)
library(plyr)
getwd()
# Read data, make validation set
train_data = read_csv("../input/train.csv")
set.seed(1992); val_inds = sample(1:nrow(train_data), floor(nrow(train_data)*0.08),F)
val_data = train_data[val_inds,]
train_data = train_data[-val_inds,]
#qr_inds = sample(1:nrow(train_data), floor(nrow(train_data)*0.004),F)
#train_data = train_data[qr_inds,]
test_data = read_csv("../input/test.csv")
#tes_inds = sample(1:nrow(train_data), floor(nrow(train_data)*0.001),F)
#test_data = test_data[tes_inds,]

print("1 done")
#train_data$question_text <- train_data$question_text  %>%  str_replace_all("\\d", " number ")
#val_data$question_text <- val_data$question_text  %>%  str_replace_all("\\d", " number ")
#test_data$question_text <- test_data$question_text  %>%  str_replace_all("\\d", " number ")
print("1 done")
#Record batch size, max words, other parameters
b_size = 512
max_words = 90*1000
maxl = 70
train.len <- nrow(train_data) #length of train set
val.len <- nrow(val_data) #length of validation set
test.len <- nrow(test_data) #length of test set
combined.data <- bind_rows(train_data, val_data,test_data);

rm(train_data,val_data,test_data)
combined.data$question_text <- str_to_lower(combined.data$question_text) %>%
  str_replace_all("\\d", " number ")

combined.data$question_text <- str_replace_all(combined.data$question_text,"\\,", " comma ")
combined.data$question_text <- str_replace_all(combined.data$question_text,"\\!", " exclamation ")
combined.data$question_text <- str_replace_all(combined.data$question_text,"\\-", " hyphen ")
combined.data$question_text <- str_replace_all(combined.data$question_text,"\\?", " question mark ")
combined.data$question_text <- str_replace_all(combined.data$question_text,"\\.", " period ")

#####
past_simple = c("arose", "awoke"," was","were","bore","beat","became","began","bent","bet","bound","bit","bled","blew","broke","bred","brought","broadcast","built",
                "burnt","burned","burst","bought","could","caught","chose","clung","came","cost","crept","cut","dealt","dug","did","drew","dreamt","dreamed","drank","drove","ate","fell",
                "fed","felt","fought","found","flew","forbade","forgot","forgave","froze","got","gave","went","ground","grew","hung","had","heard","hid","hit","held","hurt","kept","knelt","knew",
                "laid","led","leant","leaned","learnt","learned","left","lent","lay","lied","lit","lighted","lost","made","might","meant","met","mowed","had to","overtook","paid","put","read",
                "rode","rang","rose","ran","sawed","said","saw","sold","sent","set","sewed","shook","should","shed","shone","shot","showed","shrank","shut","sang","sank",
                "sat","slept","slid","smelt","sowed","spoke","spelt","spelled","spent","spilt","spilled","spat","spread","stood","stole","stuck","stung","stank","struck","swore","swept","swelled",
                "swam","swung","took","taught","tore","told","thought","threw","understood","woke","wore","wept","would","won","wound","wrote")
present_tense <- c("arise","awake","be","bear","beat","become","begin","bend","bet","bind","bite","bleed","blow","break","breed","bring","broadcast","build","burn","burst","buy","can","catch","choose","cling","come","cost","creep","cut","deal","dig","do","draw","dream","drink","drive","eat","fall","feed","feel","fight","find","fly","forbid","forget","forgive","freeze","get","give","go","grind","grow","hang","have","hear","hide","hit","hold","hurt","keep","kneel","know","lay","lead","lean","learn","leave","lent","lie","light","lose","make","may","mean","meet","mow","must","overtake","pay","put","read","ride","ring","rise","run","saw","say","see","sell","send","set","sew","shake","shall","shed","shine","shoot","show","shrink","shut","sing","sink","sit","sleep","slide","smell","sow","speak","spell","spend","spill","spit","spread","stand","steal","stick","sting","stink","strike","swear","sweep","swell","swim","swing","take","teach","tear","tell","think","throw","understand","wake","wear","weep","will","win","wind","write")
past_participate = c("arisen","awoken","been","born","borne","beaten","become","begun","bent","bet","bound","bitten","bled","blown","broken","bred","brought","broadcast","built","burnt","burned","burst","bought","been able","caught","chosen","clung","come","cost","crept","cut","dealt","dug","done","drawn","dreamt","dreamed","drunk","driven","eaten","fallen","fed","felt","fought","found","flown","forbidden","forgotten","forgiven","frozen","got","given","gone","ground","grown","hung","had","heard","hidden","hit","held","hurt","kept","knelt","known","laid","led","leant","leaned","learnt","learned","left","lent","lain","lied","lit","lighted","lost","made","meant","met","mown","mowed","overtaken","paid","put","read","ridden","rung","risen","run","sawn","sawed","said","seen","sold","sent","set","sewn","sewed","shaken","shed","shone","shot","shown","shrunk","shut","sung","sunk","sat","slept","slid","smelt","sown","sowed","spoken","spelt","spelled","spent","spilt","spilled","spat","spread","stood","stolen","stuck","stung","stunk","struck","sworn","swept","swollen","swelled","swum","swung","taken","taught","torn","told","thought","thrown","understood","woken","worn","wept","won","wound","written")
frequency_adverb <- c("always","usually","often","normally","occasionally ","sometimes","seldom","never","hardly","ever","constantly","continually","frequently","infrequently","intermittently","periodically","rarely","regularly","generally","now and then","almost never","eventually","quarterly","weekly","later","then")
time_adverb <- c("now","then","today","tomorrow","tonight","yesterday","annually","daily","fortnightly","hourly","monthly","nightly","quarterly","weekly","yearly","always","constantly","ever","frequently","generally","infrequently","never","normally","occasionally","often","rarely","regularly","seldom","sometimes","regularly","usually","already","before","early","earlier","eventually","finally","first","formerly","just","last","late","later","lately","next","previously","recently","since","soon","still","yet")
place_adverb <- c("about","above","abroad","anywhere","away","back","backwards","behind","below","down","downstairs","east","west","north","south","elsewhere","far","here","in","indoors","inside","near","nearby","off","on","out","outside","over","there","towards","under","up","upstairs","where","everywhere","somewhere","nowhere","somewhere","eastward","westward","eastwards","westwards","forwards","homewards","upwards")
manner_adverb <- c("accidentally","angrily","anxiously","awkwardly","badly","beautifully","blindly","boldly","bravely","brightly","busily","calmly","carefully","carelessly","cautiously","cheerfully","clearly","closely","correctly","courageously","cruelly","daringly","deliberately","doubtfully","eagerly","easily","elegantly","enormously","enthusiastically","equally","eventually","exactly","faithfully","fast","fatally","fiercely","fondly","foolishly","fortunately","frankly","frantically","generously","gently","gladly","gracefully","greedily","happily","hard","hastily","healthily","honestly","hungrily","hurriedly","inadequately","ingeniously","innocently","inquisitively","irritably","joyously","justly","kindly","lazily","loosely","loudly","madly","mortally","mysteriously","neatly","nervously","noisily","obediently","openly","painfully","patiently","perfectly","politely","poorly","powerfully","promptly","punctually","quickly","quietly","rapidly","rarely","really","recklessly","regularly","reluctantly","repeatedly","rightfully","roughly","rudely","sadly","safely","selfishly","sensibly","seriously","sharply","shyly","silently","sleepily","slowly","smoothly","so","softly","solemnly","speedily","stealthily","sternly","straight","stupidly","successfully","suddenly","suspiciously","swiftly","tenderly","tensely","thoughtfully","tightly","truthfully","unexpectedly","victoriously","violently","vivaciously","warmly","weakly","wearily","well","wildly","wisely")
degree_adverb <- c("extremely","terribly","amazingly","wonderfully","insanely","especially","particularly","uncommonly","unusually","remarkably","quite","pretty","rather","fairly","not especially","not particularly","never","rarely","not only","scarcely","seldom","very","too","enough","just","almost")
Noun_of_action_suffix <- c("ise","ize","ism","ist")
possessive_noun <- c("s","s")
Derivational_suffix <- c("ness","less","ity","ion", "ify")
firstperson_pronoun <- c(" i ", " me " ,"we", "us")
secondperson_pronoun <- c("you")
thirdperson_pronoun <- c("she","her","he","him","it","they","them")
relativepronoun <- c("that","which","who","whom","whose","whichever","whoever","whatever")
demonstrative_pronoun <- c("this","that","these","those")
indefinite_pronoun <- c("anybody","anyone","anything","each","either","everybody","everyone","everything","neither","nobody","no one","nothing","one","somebody","someone","something","both","few","many","several","all","any","most","none","some")
reflexive_pronoun <- c("self","selves")
interrogative_pronoun <- c("what","who","where","when","whose","which","whom")
possessive_pronoun <- c("my","mine","your","yours","his","her","its","our","their","ours","theirs","his","hers")
subject_pronoun <- c(" i ", "you","she","he","it","we","you","they")
object_pronoun <- c("you","me","her","him","it","us","them")
nominalization <- c("tion", "ment", "ness", "ity")
gerunds <- c("ing")
cojunct <- c("after","although","as","as if","as long as","as much as","as soon as","as though","because","before","by the time","even if","even though","I if","in order that","in case","lest","once","only if","provided that","unless","until","when","whenever","where","wherever","while","and","or","either","neither","nor","not only","but also","whether")
hedges <- c("usually","generally","relatively","almost","at least","nearly","roughly","typically","potentially","ultimately","around","approximately","seems to","for the most part","more or less","on average","nearly","in the neighborhood of","upwards of")
inflators <- c("very","highly","extremely","literally","truly","really","totally","greatly","key","immediately","suddenly","precisely","absolutely","intrinsically","very important to note","specific key concept")
discourse_marker <- c("anyway","like","right","you know","fine","now","so","I mean","good","oh","well","as I say","great","okay","mind you","for a start","firstly","in addition","moreover","on the other hand","secondly","in conclusion","on the one hand","to begin with","thirdly","in sum ")
modal_possiblity <- c("can", "may", "might", "could")
modal_necessity <- c("ought","should","must")
modal_predictive <- c("will","would","shall")
verb_seem <- c("seem")
public_verb <- c("affirm", "announce", "boast", "confirm", "declare","state","confirm","report","sai","interview","talk","mention")
private_verb <- c("think","deduce","assume", "believe", "doubt", "know","fear","guess","like","remember","wish","hope","forget","underst","wonder","feel","love","hate")
power_verbs <- c("abolish","accelerate","achieve","act","adopt","align","anticipate","apply","assess","avoid","boost","break","bridge","build","burn","capture","change","choose","clarify","clobber","confront","connect","conquer","convert","create","decide","define","defuse","deliver","deploy","design","develop","diagnose","discover","drive","eliminate","ensure","establish","evaluate","exploit","explore","filter","finalize","find","focus","foresee","gain","gather","generate","grasp","identify","ignite","implement","improve","increase","innovate","inspire","intensify","lead","learn","leverage","manage","master","maximize","measure","mobilize","motivate","overcome","penetrate","persuade","plan","position","prepare","prevent","profit","raise","reconsider","reduce","refresh","replace","resist","respond","retain","save","scan","shatter","shadeoff","sidestep","simplify","slash","solve","stimulate","stop","stretch","succeed","supplement","take","transfer","transform","understand","unleash","unravel","use","win")
General_swear_words <- c("twat","tits","son of a bitch","sod off","snatch","shit","pussy","punani","prick","piss","munter","mofo","fuck","minge","knob","damn","ginger","gash","flaps","eff","feck","dick","cunt","crap","cow","clunge","bull","bugger","bollocks","bitch","beaver","ass","arse","bastard","balls")
Sexual_references <- c("bonk","bukkake","cocksucker","dildo","ho","slut","jizz","nonce","prickteaser","rape","shag","skank","slag","skank","slapper","tart","wanker","whore")
Age_dsicrim <- c("coffin dodger","fop","old bag")
Religion_racist <- c("fenian","kafir","kufaar","kike","papist","prod","taig","yid","muslim","arab","dalit","caste","sect","jihad")
orientation_identity <- c("batty boy","bender","bum boy","bumclat","bummer","chi-chi","dyke","fag","fudge","gay","homo","lesbo","lesb","lezza","muff","pansy","poof","queer","rugmuncher","tranny")
mental_physicalhealth <- c("cretin","cripple","div","loony","mental","midget","mong","nutter","psycho","retard","schizo","spakka","spaz","spastic")
Google_banned_badwords <- c("4r5e","5h1t","5hit","a55","anal","anus","ar5e","arrse","arse","ass","ass-fucker","asses","assfucker","assfukka","asshole","assholes","asswhole","a_s_s","b!tch","b00bs","b17ch","b1tch","ballbag","balls","ballsack","bastard","beastial","beastiality","bellend","bestial","bestiality","bi+ch","biatch","bitch","bitcher","bitchers","bitches","bitchin","bitching","bloody","blow job","blowjob","blowjobs","boiolas","bollock","bollok","boner","boob","boobs","booobs","boooobs","booooobs","booooooobs","breasts","buceta","bugger","bum","bunny fucker","butt","butthole","buttmuch","buttplug","c0ck","c0cksucker","carpet muncher","cawk","chink","cipa","cl1t","clit","clitoris","clits","cnut","cock","cock-sucker","cockface","cockhead","cockmunch","cockmuncher","cocks","cocksuck ","cocksucked ","cocksucker","cocksucking","cocksucks ","cocksuka","cocksukka","cok","cokmuncher","coksucka","coon","cox","crap","cum","cummer","cumming","cums","cumshot","cunilingus","cunillingus","cunnilingus","cunt","cuntlick ","cuntlicker ","cuntlicking ","cunts","cyalis","cyberfuc","cyberfuck ","cyberfucked ","cyberfucker","cyberfuckers","cyberfucking ","d1ck","damn","dick","dickhead","dildo","dildos","dink","dinks","dirsa","dlck","dog-fucker","doggin","dogging","donkeyribber","doosh","duche","dyke","ejaculate","ejaculated","ejaculates ","ejaculating ","ejaculatings","ejaculation","ejakulate","f u c k","f u c k e r","f4nny","fag","fagging","faggitt","faggot","faggs","fagot","fagots","fags","fanny","fannyflaps","fannyfucker","fanyy","fatass","fcuk","fcuker","fcuking","feck","fecker","felching","fellate","fellatio","fingerfuck ","fingerfucked ","fingerfucker ","fingerfuckers","fingerfucking ","fingerfucks ","fistfuck","fistfucked ","fistfucker ","fistfuckers ","fistfucking ","fistfuckings ","fistfucks ","flange","fook","fooker","fuck","fucka","fucked","fucker","fuckers","fuckhead","fuckheads","fuckin","fucking","fuckings","fuckingshitmotherfucker","fuckme ","fucks","fuckwhit","fuckwit","fudge packer","fudgepacker","fuk","fuker","fukker","fukkin","fuks","fukwhit","fukwit","fux","fux0r","f_u_c_k","gangbang","gangbanged ","gangbangs ","gaylord","gaysex","goatse","God","god-dam","god-damned","goddamn","goddamned","hardcoresex ","hell","heshe","hoar","hoare","hoer","homo","hore","horniest","horny","hotsex","jack-off ","jackoff","jap","jerk-off ","jism","jiz ","jizm ","jizz","kawk","knob","knobead","knobed","knobend","knobhead","knobjocky","knobjokey","kock","kondum","kondums","kum","kummer","kumming","kums","kunilingus","l3i+ch","l3itch","labia","lmfao","lust","lusting","m0f0","m0fo","m45terbate","ma5terb8","ma5terbate","masochist","master-bate","masterb8","masterbat*","masterbat3","masterbate","masterbation","masterbations","masturbate","mo-fo","mof0","mofo","mothafuck","mothafucka","mothafuckas","mothafuckaz","mothafucked ","mothafucker","mothafuckers","mothafuckin","mothafucking ","mothafuckings","mothafucks","mother fucker","motherfuck","motherfucked","motherfucker","motherfuckers","motherfuckin","motherfucking","motherfuckings","motherfuckka","motherfucks","muff","mutha","muthafecker","muthafuckker","muther","mutherfucker","n1gga","n1gger","nazi","nigg3r","nigg4h","nigga","niggah","niggas","niggaz","nigger",
                            "niggers ","nob","nob jokey","nobhead","nobjocky","nobjokey","numbnuts","nutsack","orgasim ","orgasims ","orgasm","orgasms ","p0rn","pawn","pecker","penis","penisfucker","phonesex","phuck","phuk","phuked","phuking","phukked","phukking","phuks","phuq","pigfucker","pimpis","piss","pissed","pisser","pissers","pisses ","pissflaps","pissin ","pissing","pissoff ","poop","porn","porno","pornography","pornos","prick","pricks ","pron","pube","pusse","pussi","pussies","pussy","pussys ","rectum","retard","rimjaw","rimming","s hit","s.o.b.","sadist","schlong","screwing","scroat","scrote","scrotum","semen","sex","sh!+","sh!t","sh1t","shag","shagger","shaggin","shagging","shemale","shi+","shit","shitdick","shite","shited","shitey","shitfuck","shitfull","shithead","shiting","shitings","shits","shitted","shitter","shitters ","shitting","shittings","shitty ","skank","slut","sluts","smegma","smut","snatch","son-of-a-bitch","spac","spunk","s_h_i_t","t1tt1e5","t1tties","teets","teez","testical","testicle","tit","titfuck","tits","titt","tittie5","tittiefucker","titties","tittyfuck","tittywank","titwank","tosser","turd","tw4t","twat","twathead","twatty","twunt","twunter","v14gra","v1gra","vagina","viagra","vulva","w00se","wang","wank","wanker","wanky","whoar","whore","willies","willy","xrated","xxx")
bad_words <- c("abbo","abo","abortion","abuse","addict","addicts","adult","africa","african",
               "alla","allah","alligatorbait","amateur","american","anal","analannie","analsex",
               "angie","angry","anus","arab","arabs","areola","argie","aroused","arse","arsehole",
               "asian","ass","assassin","assassinate","assassination","assault","assbagger","assblaster",
               "assclown","asscowboy", "asses","assfuck","assfucker","asshat","asshole","assholes","asshore",
               "assjockey","asskiss","asskisser","assklown","asslick","asslicker","asslover","assman","assmonkey",
               "assmunch","assmuncher","asspacker","asspirate","asspuppies","assranger","asswhore","asswipe",
               "athletesfoot","attack","australian","babe","babies","backdoor","backdoorman","backseat",
               "badfuck","balllicker","balls","ballsack","banging","baptist","barelylegal","barf","barface",
               "barfface","bast","bastard","bazongas","bazooms","beaner","beast","beastality","beastial",
               "beastiality","beatoff","beat-off","beatyourmeat","beaver","bestial","bestiality","bi","biatch",
               "bible","bicurious","bigass","bigbastard","bigbutt","bigger","bisexual","bi-sexual","bitch",
               "bitcher","bitches","bitchez","bitchin","bitching","bitchslap","bitchy","biteme","black",
               "blackman","blackout","blacks","blind","blow","blowjob","boang","bogan","bohunk","bollick",
               "bollock","bomb","bombers","bombing","bombs","bomd","bondage","boner","bong","boob","boobies",
               "boobs","booby","boody","boom","boong","boonga","boonie","booty","bootycall","bountybar","bra",
               "brea5t","breast","breastjob","breastlover","breastman","brothel","bugger","buggered","buggery",
               "bullcrap","bulldike","bulldyke","bullshit","bumblefuck","bumfuck","bunga","bunghole","buried",
               "burn","butchbabes","butchdike","butchdyke","butt","buttbang","butt-bang","buttface","buttfuck",
               "butt-fuck","buttfucker","butt-fucker","buttfuckers","butt-fuckers","butthead","buttman",
               "buttmunch","buttmuncher","buttpirate","buttplug","buttstain","byatch","cacker","cameljockey",
               "cameltoe","canadian","cancer","carpetmuncher","carruth","catholic","catholics","cemetery","chav",
               "cherrypopper","chickslick","chin","chinaman","chinamen","chinese","chink","chinky","choad",
               "chode","christ","christian","church","cigarette","cigs","clamdigger","clamdiver","clit",
               "clitoris","clogwog","cocaine","cock","cockblock","cockblocker","cockcowboy","cockfight",
               "cockhead","cockknob","cocklicker","cocklover","cocknob","cockqueen","cockrider","cocksman",
               "cocksmith","cocksmoker","cocksucer","cocksuck ","cocksucked ","cocksucker","cocksucking",
               "cocktail","cocktease","cocky","cohee","coitus","color","colored","coloured","commie","communist",
               "condom","conservative","conspiracy","coolie","cooly","coon","coondog","copulate","cornhole",
               "corruption","cra5h","crabs","crack","crackpipe","crackwhore","crack-whore","crap","crapola",
               "crapper","crappy","crash","creamy","crime","crimes","criminal","criminals","crotch",
               "crotchjockey","crotchmonkey","crotchrot","cum","cumbubble","cumfest","cumjockey","cumm",
               "cummer","cumming","cumquat","cumqueen","cumshot","cunilingus","cunillingus","cunn","cunnilingus",
               "cunntt","cunt","cunteyed","cuntfuck","cuntfucker","cuntlick","cuntlicker","cuntlicking",
               "cuntsucker","cybersex","cyberslimer","dago","dahmer","dammit","damn","damnation","damnit",
               "darkie","darky","datnigga","dead","deapthroat","death","deepthroat","defecate","dego","demon",
               "deposit","desire","destroy","deth","devil","devilworshipper","dick","dickbrain","dickforbrains","dickhead",
               "dickless","dicklick","dicklicker","dickman","dickwad","dickweed","diddle","die","died","dies",
               "dike","dildo","dingleberry","dink","dipshit","dipstick","dirty","disease","diseases","disturbed",
               "dive","dix","dixiedike","dixiedyke","doggiestyle","doggystyle","dong","doodoo","doo-doo","doom",
               "dope","dragqueen","dragqween","dripdick","drug","drunk","drunken","dumb","dumbass","dumbbitch",
               "dumbfuck","dyefly","dyke","easyslut","eatballs","eatme","eatpussy","ecstacy","ejaculate",
               "ejaculated","ejaculating","ejaculation","enema","enemy","erect","erection","ero","escort",
               "ethiopian","ethnic","european","evl","excrement","execute","executed","execution","executioner",
               "explosion","facefucker","faeces","fag","fagging","faggot","fagot","failed","failure","fairies",
               "fairy","faith","fannyfucker","fart","farted","farting ","farty ","fastfuck","fat","fatah","fatass",
               "fatfuck","fatfucker","fatso","fckcum","fear","feces","felatio ","felch","felcher","felching","fellatio",
               "feltch","feltcher","feltching","fetish","fight","filipina","filipino","fingerfood","fingerfuck ","fingerfucked",
               "fingerfucker ","fingerfuckers","fingerfucking ","fire","firing","fister","fistfuck","fistfucked","fistfucker","fistfucking",
               "fisting","flange","flasher","flatulence","floo","flydie","flydye","fok","fondle","footaction","footfuck","footfucker","footlicker",
               "footstar","fore","foreskin","forni","fornicate","foursome","fourtwenty","fraud","freakfuck","freakyfucker","freefuck","fu","fubar",
               "fuc","fucck","fuck","fucka","fuckable","fuckbag","fuckbuddy","fucked","fuckedup","fucker","fuckers","fuckface","fuckfest","fuckfreak",
               "fuckfriend","fuckhead","fuckher","fuckin","fuckina","fucking","fuckingbitch","fuckinnuts","fuckinright","fuckit","fuckknob","fuckme ",
               "fuckmehard","fuckmonkey","fuckoff","fuckpig","fucks","fucktard","fuckwhore","fuckyou","fudgepacker","fugly","fuk","fuks","funeral",
               "funfuck","fungus","fuuck","gangbang","gangbanged ","gangbanger","gangsta","gatorbait","gay","gaymuthafuckinwhore","gaysex ","geez",
               "geezer","geni","genital","german","getiton","gin","ginzo","gipp","girls","givehead","glazeddonut","gob","god","godammit","goddamit",
               "goddammit","goddamn","goddamned","goddamnes","goddamnit","goddamnmuthafucker","goldenshower","gonorrehea","gonzagas","gook","gotohell",
               "goy","goyim","greaseball","gringo","groe","gross","grostulation","gubba","gummer","gun","gyp","gypo","gypp","gyppie","gyppo","gyppy",
               "hamas","handjob","hapa","harder","hardon","harem","headfuck","headlights","hebe","heeb","hell","henhouse","heroin","herpes","heterosexual",
               "hijack","hijacker","hijacking","hillbillies","hindoo","hiscock","hitler","hitlerism","hitlerist","hiv","ho","hobo","hodgie","hoes","hole",
               "holestuffer","homicide","homo","homobangers","homosexual","honger","honk","honkers","honkey","honky","hook","hooker","hookers","hooters",
               "hore","hork","horn","horney","horniest","horny","horseshit","hosejob","hoser","hostage","hotdamn","hotpussy","hottotrot","hummer","husky",
               "hussy","hustler","hymen","hymie","iblowu","idiot","ikey","illegal","incest","insest","intercourse","interracial","intheass","inthebuff",
               "israel","israeli","italiano","itch","jackass","jackoff","jackshit","jacktheripper","jade","jap","japanese","japcrap","jebus","jeez",
               "jerkoff","jesus","jesuschrist","jew","jewish","jiga","jigaboo","jigg","jigga","jiggabo","jigger ","jiggy","jihad","jijjiboo",
               "jimfish","jism","jiz ","jizim","jizjuice","jizm ","jizz","jizzim","jizzum","joint","juggalo","jugs","junglebunny","kaffer",
               "kaffir","kaffre","kafir","kanake","kid","kigger","kike","kill","killed","killer","killing","kills","kink","kinky","kissass",
               "kkk","knife","knockers","kock","kondum","koon","kotex","krap","krappy","kraut","kum","kumbubble","kumbullbe","kummer","kumming",
               "kumquat","kums","kunilingus","kunnilingus","kunt","ky","kyke","lactate","laid","lapdance","latin","lesbain","lesbayn","lesbian",
               "lesbin","lesbo","lez","lezbe","lezbefriends","lezbo","lezz","lezzo","liberal","libido","licker","lickme","lies","limey","limpdick",
               "limy","lingerie","liquor","livesex","loadedgun","lolita","looser","loser","lotion","lovebone","lovegoo","lovegun","lovejuice","lovemuscle",
               "lovepistol","loverocket","lowlife","lsd","lubejob","lucifer","luckycammeltoe","lugan","lynch","macaca","mad","mafia","magicwand","mams","manhater",
               "manpaste","marijuana","mastabate","mastabater","masterbate","masterblaster","mastrabator","masturbate","masturbating","mattressprincess","meatbeatter",
               "meatrack","meth","mexican","mgger","mggor","mickeyfinn","mideast","milf","minority","mockey","mockie","mocky","mofo","moky","moles","molest","molestation",
               "molester","molestor","moneyshot","mooncricket","mormon","moron","moslem","mosshead","mothafuck","mothafucka","mothafuckaz","mothafucked ","mothafucker","mothafuckin",
               "mothafucking ","mothafuckings","motherfuck","motherfucked","motherfucker","motherfuckin","motherfucking","motherfuckings","motherlovebone","muff","muffdive","muffdiver",
               "muffindiver","mufflikcer","mulatto","muncher","munt","murder","murderer","muslim","naked","narcotic","nasty","nastybitch","nastyho","nastyslut","nastywhore","nazi","necro",
               "negro","negroes","negroid","negro","nig","niger","nigerian","nigerians","nigg","nigga","niggah","niggaracci","niggard","niggarded","niggarding","niggardliness","niggardlinesss",
               "niggardly","niggards","niggard","niggaz","nigger","niggerhead","niggerhole","niggers","niggers","niggle","niggled","niggles","niggling","nigglings","niggor","niggur","niglet",
               "nignog","nigr","nigra","nigre","nip","nipple","nipplering","nittit","nlgger","nlggor","nofuckingway","nook","nookey","nookie","noonan","nooner","nude","nudger","nuke","nutfucker",
               "nymph","ontherag","oral","orga","orgasim ","orgasm","orgies","orgy","osama","paki","palesimian",
               "palestinian","pansies","pansy","panti","panties","payo","pearlnecklace","peck","pecker","peckerwood",
               "pee","peehole","pee-pee","peepshow","peepshpw","pendy","penetration","peni5","penile","penis","penises",
               "penthouse","period","perv","phonesex","phuk","phuked","phuking","phukked","phukking","phungky","phuq",
               "pi55","picaninny","piccaninny","pickaninny","piker","pikey","piky","pimp","pimped","pimper","pimpjuic",
               "pimpjuice","pimpsimp","pindick","piss","pissed","pisser","pisses ","pisshead","pissin ","pissing",
               "pissoff ","pistol","pixie","pixy","playboy","playgirl","pocha","pocho","pocketpool","pohm","polack",
               "pom","pommie","pommy","poo","poon","poontang","poop","pooper","pooperscooper","pooping","poorwhitetrash",
               "popimp","porchmonkey","porn","pornflick","pornking","porno","pornography","pornprincess","pot","poverty",
               "premature","pric","prick","prickhead","primetime","propaganda","pros","prostitute","protestant","pu55i",
               "pu55y","pube","pubic","pubiclice","pud","pudboy","pudd","puddboy","puke","puntang","purinapricness","puss",
               "pussie","pussies","pussy","pussycat","pussyeater","pussyfucker","pussylicker","pussylips","pussylover",
               "pussypounder","pusy","quashie","queef","queer","quickie","quim","ra8s","rabbi","racial","racist","radical",
               "radicals","raghead","randy","rape","raped","raper","rapist","rearend","rearentry","rectum","redlight",
               "redneck","reefer","reestie","refugee","reject","remains","rentafuck","republican","rere","retard",
               "retarded","ribbed","rigger","rimjob","rimming","roach","robber","roundeye","rump","russki","russkie",
               "sadis","sadom","samckdaddy","sandm","sandnigger","satan","scag","scallywag","scat","schlong","screw",
               "screwyou","scrotum","scum","semen","seppo","servant","sex","sexed","sexfarm","sexhound","sexhouse",
               "sexing","sexkitten","sexpot","sexslave","sextogo","sextoy","sextoys","sexual","sexually","sexwhore",
               "sexy","sexymoma","sexy-slim","shag","shaggin","shagging","shat","shav","shawtypimp","sheeney","shhit",
               "shinola","shit","shitcan","shitdick","shite","shiteater","shited","shitface","shitfaced","shitfit",
               "shitforbrains","shitfuck","shitfucker","shitfull","shithapens","shithappens","shithead","shithouse",
               "shiting","shitlist","shitola","shitoutofluck","shits","shitstain","shitted","shitter","shitting",
               "shitty ","shoot","shooting","shortfuck","showtime","sick","sissy","sixsixsix","sixtynine","sixtyniner",
               "skank","skankbitch","skankfuck","skankwhore","skanky","skankybitch","skankywhore","skinflute","skum",
               "skumbag","slant","slanteye","slapper","slaughter","slav","slave","slavedriver","sleezebag",
               "sleezeball","slideitin","slime","slimeball","slimebucket","slopehead","slopey","slopy","slut",
               "sluts","slutt","slutting","slutty","slutwear","slutwhore","smack","smackthemonkey","smut",
               "snatch","snatchpatch","snigger","sniggered","sniggering","sniggers","sniggers","sniper",
               "snot","snowback","snownigger","sob","sodom","sodomise","sodomite","sodomize","sodomy",
               "sonofabitch","sonofbitch","sooty","sos","soviet","spaghettibender","spaghettinigger",
               "spank","spankthemonkey","sperm","spermacide","spermbag","spermhearder","spermherder",
               "spic","spick","spig","spigotty","spik","spit","spitter","splittail","spooge","spreadeagle",
               "spunk","spunky","squaw","stagg","stiffy","strapon","stringer","stripclub","stroke","stroking","stupid",
               "stupidfuck","stupidfucker","suck","suckdick","sucker","suckme","suckmyass","suckmydick","suckmytit",
               "suckoff","suicide","swallow","swallower","swalow","swastika","sweetness","syphilis","taboo","taff",
               "tampon","tang","tantra","tarbaby","tard","teat","terror","terrorist","teste","testicle","testicles",
               "thicklips","thirdeye","thirdleg","threesome","threeway","timbernigger","tinkle","tit","titbitnipply",
               "titfuck","titfucker","titfuckin","titjob","titlicker","titlover","tits","tittie","titties","titty",
               "tnt","toilet","tongethruster","tongue","tonguethrust","tonguetramp","tortur","torture","tosser",
               "towelhead","trailertrash","tramp","trannie","tranny","transexual","transsexual","transvestite",
               "triplex","trisexual","trojan","trots","tuckahoe","tunneloflove","turd","turnon","twat","twink",
               "twinkie","twobitwhore","uck","uk","unfuckable","upskirt","uptheass","upthebutt","urinary","urinate",
               "urine","usama","uterus","vagina","vaginal","vatican","vibr","vibrater","vibrator","vietcong","violence",
               "virgin","virginbreaker","vomit","vulva","wab","wank","wanker","wanking","waysted","weapon","weenie","weewee",
               "welcher","welfare","wetb","wetback","wetspot","whacker","whash","whigger","whiskey","whiskeydick","whiskydick",
               "whit","whitenigger","whites","whitetrash","whitey","whiz","whop","whore","whorefucker","whorehouse","wigger","willie",
               "williewanker","willy","wn","wog","womens","wop","wtf","wuss","wuzzie","xtc","xxx","yankee","yellowman","zigabo","zipperhead")
Race_ethnicity <- c("chink","choc ice","colored","coloured","coon","darky","dago","gippo","golliwog","jock","honky","hun","jap","kraut","nazi","negro","nigg","paki","pikey","polack","raghead","sambo","slope","spic","taff","wog","wop")
country_nationality <- c("China","Armenia","Argentina","Afghanistan","Albania","United Arab Emirates","Austria","Spain","France","United States","Germany","Brazil","Switzerland","Côte d'Ivoire","Chile","Romania","Portugal","Paraguay","Réunion",
                         "Poland","Uruguay","Uzbekistan","Venezuela, Bolivarian Rep. of","Viet Nam","Vanuatu","Afghan",
                         "Afghanistan","Afrikaans","Albania","Albanian","Algeria","Algerian","American","American Samoa",
                         "Amharic","Andorra","Angola","Anguilla","Antigua and Barbuda","Arabic","Arabic, Kurdish","Arabiv",
                         "Argentina","Argentine","Argentinian","Armenia","Aruba","Australia","Australian","Austria",
                         "Austrian","Azerbaijan","Bahamas","Bahrain","Bangladesh","Bangladeshi","Barbados","Batswana",
                         "Belarus","Belgian","Belgium","Belize","Bengali","Benin","Bermuda","Bhutan","Bolivia","Bolivian",
                         "Bosnia and Herzegovina","Botswana","Brazil","Brazilian","British","British Virgin Islands",
                         "Brunei Darussalam","Bulgaria","Bulgarian","Burkina Faso","Burundi","Cambodia","Cambodian",
                         "Cameroon","Cameroonian","Canada","Canadian","Cape Verde","Cayman Islands",
                         "Central African Republic","Chad","Chile","Chilean","China","Chinese","Colombia","Colombian",
                         "Comoros","Congo","Congo, Democratic Republic of the","Cook Islands","Costa Rica","Costa Rican",
                         "Côte d'Ivoire","Country","Croatia","Croatian","Cuba","Cuban","Cyprus","Czech","Czech Republic",
                         "Danish","Denmark","Djibouti","Dominica","Dominican","Dominican Republic","Dutch","Ecuador",
                         "Ecuadorian","Egypt","Egyptian","El Salvador","Emirati","English","English ",
                         "Equatorial Guinea","Eritrea","Estonia","Estonian","Ethiopia","Ethiopian",
                         "Faeroe Islands","Falkland Islands (Malvinas)","Fiji","Fijian","Finland","Finnish",
                         "Flemish","France","French","French Guiana","French Polynesia","Gabon","Gambia",
                         "Georgia","German","Germany","Ghana","Ghanaian","Gibraltar","Greece","Greek","Greenland",
                         "Grenada","Guadeloupe","Guam","Guatemala","Guatemalan","Guernsey","Guinea","Guinea-Bissau",
                         "Guyana","Haiti","Haitian","Hebrew","Hindi","Honduran","Honduras","Hong Kong","Hungarian",
                         "Hungary","Iceland","Icelandic","India","Indian","Indonesia","Indonesian","Iran","Iranian",
                         "Iraq","Iraqi","Ireland","Irish","Isle of Man","Israel","Israeli","Italian","Italy","Jamaica",
                         "Jamaican","Japan","Japanese","Jersey","Jordan","Jordanian","Kazakhstan","Kenya","Kenyan",
                         "Kiribati","Korea, Dem. People's Rep. of","Korea, Republic of","Korean","Kuwait","Kuwaiti",
                         "Kyrgyzstan","Lao","Lao People's Dem. Rep.","Laotian","Latvia","Latvian","Lebanese","Lebanon",
                         "Lesotho","Liberia","Libyan","Libyan Arab Jamahiriya","Liechtenstein","Lithuania","Lithuanian",
                         "Luxembourg","Macau, China","Macedonia, The former Yugoslav Rep. of","Madagascar",
                         "Malawi","Malay","Malaysia","Malaysian","Maldives","Mali","Malian","Malta","Maltese",
                         "Marshall Islands","Martinique","Mauritania","Mauritius","Mexican","Mexico","Moldova",
                         "Monaco","Mongolia","Mongolian","Montenegro","Montserrat","Moroccan","Morocco","Mozambican",
                         "Mozambique","Myanmar","Namibia","Namibian","Nauru","Nepal","Nepalese","Nepali","Netherlands",
                         "New Caledonia","New Zealand","Nicaragua","Nicaraguan","Niger","Nigeria","Nigerian","Niue",
                         "Norfolk Island","Northern Mariana Islands","Norway","Norwegian","Oman","Pakistan","Pakistani",
                         "Palau","Panama","Panamanian","Papua New Guinea","Paraguay","Paraguayan","Pashto","Persian",
                         "Peru","Peruvian","Philippine","Philippines","Poland","Polish","Portugal","Portuguese",
                         "Puerto Rico","Qatar","Réunion","Romania","Romanian","Russian","Russian Federation","Rwanda",
                         "Saint Helena","Saint Kitts and Nevis","Saint Lucia","Saint Pierre and Miquelon",
                         "Saint Vincent and the Grenadines","Salvadorian","Samoa","San Marino",
                         "Sao Tome and Principe","Saudi","Saudi Arabia","Scottish","Senegal",
                         "Senegalese","Serbia","Serbian","Seychelles","Sierra Leone","Singapore",
                         "Singaporean","Sinhala","Slovak","Slovakia","Slovenia","Solomon Islands",
                         "Somalia","Africa","South African","Sudan","Spain","Spanish","Sri Lanka",
                         "Sri Lankan","Sudan","Sudanese","Suriname","Swahili","Swaziland","Sweden",
                         "Swedish","Swiss","Switzerland","Syrian","Syrian Arab Republic","Tagalog",
                         "Taiwan, China","Taiwanese","Tajik","Tajikistan","Tajikistani","Tamil",
                         "Tanzania","Thai","Thailand","Timor-Leste","Togo","Tokelau","Tonga","Tongan",
                         "Trinidad and Tobago","Tunisia","Tunisian","Turkey","Turkish","Turkmenistan",
                         "Turks","Caicos","Tuvalu","Uganda","Ukraine","Ukrainian",
                         "United Arab Emirates","United Kingdom","United States",
                         "Urdu","Uruguay","Uruguayan","Uzbekistan","Vanuatu",
                         "Venezuela","Venezuelan","Viet Nam","Vietnamese",
                         "Virgin Islands","Wallis and Futuna Islands","Welsh",
                         "West Bank and Gaza Strip","Western Sahara","Yemen","Zambia","Zambian",
                         "Zimbabwe"," Setswana","Afghan","Albanian","Algerian","Argentine",
                         "Argentinian","Australian","Austrian","Bangladeshi","Belgian",
                         "Bolivian","Batswana","Brazilian","Bulgarian","Cambodian","Cameroonian","Canadian","Chilean",
                         "Chinese","Colombian","Costa Rican","Croatian","Cuban","Czech","Danish","Dominican","Ecuadorian",
                         "Egyptian","Salvadorian","English")
library(stringi)
library(stringr)
library(plyr)
combined.data$country_nationality <- laply(combined.data$question_text, function(question_text) sum(stri_detect_fixed(question_text, country_nationality)))
combined.data$bad_words <- laply(combined.data$question_text, function(question_text) sum(stri_detect_fixed(question_text, bad_words)))
summary(combined.data$country_nationality)
combined.data$General_swear_words =  laply(combined.data$question_text, function(question_text) sum(stri_detect_fixed(question_text, General_swear_words)))
combined.data$Google_banned_badwords <- laply(combined.data$question_text, function(question_text) sum(stri_detect_fixed(question_text, Google_banned_badwords)))
combined.data$Sexual_references =  laply(combined.data$question_text, function(question_text) sum(stri_detect_fixed(question_text, Sexual_references)))
combined.data$Age_dsicrim =  laply(combined.data$question_text, function(question_text) sum(stri_detect_fixed(question_text, Age_dsicrim)))
combined.data$Religion_racist =  laply(combined.data$question_text, function(question_text) sum(stri_detect_fixed(question_text, Religion_racist)))
combined.data$orientation_identity =  laply(combined.data$question_text, function(question_text) sum(stri_detect_fixed(question_text, orientation_identity)))
combined.data$mental_physicalhealth =  laply(combined.data$question_text, function(question_text) sum(stri_detect_fixed(question_text, mental_physicalhealth)))
combined.data$Race_ethnicity =  laply(combined.data$question_text, function(question_text) sum(stri_detect_fixed(question_text, Race_ethnicity)))
combined.data$Noun_of_action_suffix =  laply(combined.data$question_text, function(question_text) sum(stri_detect_fixed(question_text, Noun_of_action_suffix)))
combined.data$possessive_noun =  laply(combined.data$question_text, function(question_text) sum(stri_detect_fixed(question_text, possessive_noun)))
combined.data$Derivational_suffix =  laply(combined.data$question_text, function(question_text) sum(stri_detect_fixed(question_text, Derivational_suffix)))
combined.data$nominalization =  laply(combined.data$question_text, function(question_text) sum(stri_detect_fixed(question_text, nominalization)))
combined.data$public_verb =  laply(combined.data$question_text, function(question_text) sum(stri_detect_fixed(question_text, public_verb)))
combined.data$private_verb =  laply(combined.data$question_text, function(question_text) sum(stri_detect_fixed(question_text, private_verb)))
combined.data$power_verbs =  laply(combined.data$question_text, function(question_text) sum(stri_detect_fixed(question_text, power_verbs)))
combined.data <- combined.data%>%  mutate(length = str_length(question_text),
                                          ncap = str_count(question_text, "[A-Z]"),
                                          nword = str_count(question_text, "\\w+"),
                                          totalredflagwords = country_nationality+Sexual_references+Age_dsicrim+Religion_racist+orientation_identity+mental_physicalhealth+Race_ethnicity+Google_banned_badwords+bad_words,
                                          country_nationality_ratio = country_nationality/nword,
                                          bad_words_ratio = bad_words/nword,
                                          Google_banned_badwords_ratio = Google_banned_badwords/nword, 
                                          Sexual_references_ratio = Sexual_references/nword,
                                          Age_dsicrim_ratio = Age_dsicrim/nword,
                                          Race_ethnicity_ratio = Race_ethnicity/nword,
                                          mental_physicalhealth_ratio = mental_physicalhealth/nword,
                                          orientation_identity_ratio = orientation_identity/nword,
                                          Religion_racist_ratio = Religion_racist/nword,
                                          redflagtowordratio= totalredflagwords/nword,
                                          public_verb_ratio = public_verb/nword,
                                          private_verb_ratio = private_verb/nword,
                                          power_verbs_ratio = power_verbs/nword,
                                          ncap_len = ncap / length,
                                          nnumers = str_count(question_text, fixed("!")),
                                          nexcl = str_count(question_text, fixed("!")),
                                          nquest = str_count(question_text, fixed("?")),
                                          npunct = str_count(question_text, "[[:punct:]]"),
                                          ncommas = str_count(question_text, fixed(",")),
                                          avgwordlen = length/nword,
                                          Derivational_suffix_ratio = Derivational_suffix/nword,
                                          Noun_of_action_suffix_ratio = Noun_of_action_suffix/nword,
                                          nominalization_ratio = nominalization/nword,
                                          nsymb = str_count(question_text, "&|@|#|\\$|%|\\*|\\^"),
                                          nsmile = str_count(question_text, "((?::|;|=)(?:-)?(?:\\)|D|P))"),
                                          nsentence = str_count(question_text, fixed(".")))

#VERY minor clean-up
head(combined.data,10)

#Tokenize
wordseq = text_tokenizer(num_words = max_words) %>%
  fit_text_tokenizer(combined.data$question_text)
text2s = texts_to_sequences(wordseq, combined.data$question_text ) %>%
  pad_sequences( maxlen = maxl) 
word_index = wordseq$word_index
wordindex = unlist(wordseq$word_index)
rm(wordseq,val_inds); gc(reset=T)
print("start training")
#Split train/val/test sets

x_train1 = text2s[1:train.len,]
y_train = combined.data$target[1:train.len]
tr <- combined.data[1:train.len,-c(1:3)]
combined.data <- combined.data[-c(1:train.len),]; text2s <- text2s[-c(1:train.len),]

x_val1 = text2s[1:val.len,]
y_val = combined.data$target[1:val.len]
val <- combined.data[1:val.len,-c(1:3)]
combined.data <- combined.data[-c(1:val.len),]; text2s <- text2s[-c(1:val.len),]

x_test1 = text2s[1:test.len,]
te <- combined.data[1:test.len,-c(1:3)]
test_qid <- combined.data$qid
rm(combined.data,text2s)
ncol <- ncol(tr)

val %<>%  mutate_all(funs(ifelse(is.nan(.), NA, .))) %>% 
  mutate_all(funs(ifelse(is.infinite(.), NA, .))) %>%
  mutate_all(funs(ifelse(is.na(.), 0, .)))
te %<>%  mutate_all(funs(ifelse(is.nan(.), NA, .))) %>% 
  mutate_all(funs(ifelse(is.infinite(.), NA, .)))%>%
  mutate_all(funs(ifelse(is.na(.), 0, .)))
tr%<>%  mutate_all(funs(ifelse(is.nan(.), NA, .))) %>% 
  mutate_all(funs(ifelse(is.infinite(.), NA, .)))%>%
  mutate_all(funs(ifelse(is.na(.), 0, .)))
val <- data.matrix(val)
te <- data.matrix(te)
tr <- data.matrix(tr)

#Run model for each of the three embbedings
embed.paths = list("../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec",
                   "../input/embeddings/glove.840B.300d/glove.840B.300d.txt",
                   "../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt")

#maxlen <- 64
max_words <- 90000           
emb_dim <- 300

wgt = fread("../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt", data.table = FALSE,skip=1,verbose = TRUE)
colnames(wgt)[1] <- "word"

wgt = wgt %>%
  mutate(word=gsub("[[:punct:]]"," ", rm_white(word) ))


dic_words = wgt$word
dic = data.frame(word=as.character(names(wordindex)), key = wordindex,row.names = NULL) %>%
  arrange(key) %>% 
  .[1:max_words,]
dic$word <- as.character(dic$word)

w_embed = dic %>% 
  left_join(wgt)
rm(wgt,dic); invisible(gc(reset=T))

J = ncol(w_embed)
ndim = J-2

w_embed = w_embed [1:(max_words-1),3:J] %>%
  mutate_all(as.numeric) %>%
  mutate_all(round,6) %>%
  mutate_all(funs(replace(., is.na(.), 0))) 

colnames(w_embed) = paste0("V",1:ndim)
w_embed = rbind(rep(0, ndim), w_embed) %>%
  as.matrix()

w_embed = list(array(w_embed , c(max_words, ndim)))

wgt = fread("../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec", data.table = FALSE,skip=1, verbose= TRUE)
colnames(wgt)[1] <- "word"

wgt = wgt %>%
  mutate(word=gsub("[[:punct:]]"," ", rm_white(word) ))


dic_words = wgt$word
dic = data.frame(word=as.character(names(wordindex)), key = wordindex,row.names = NULL) %>%
  arrange(key) %>% 
  .[1:max_words,]
dic$word <- as.character(dic$word)

w_embed1 = dic %>% 
  left_join(wgt)
rm(wgt,dic); invisible(gc(reset=T))

J = ncol(w_embed1)
ndim2 = J-2
w_embed1 = w_embed1 [1:(max_words-1),3:J] %>%
  mutate_all(as.numeric) %>%
  mutate_all(round,6) %>%
  mutate_all(funs(replace(., is.na(.), 0))) 

colnames(w_embed1) = paste0("V",1:ndim2)
w_embed1 = rbind(rep(0, ndim2), w_embed1) %>%
  as.matrix()


w_embed1 = list(array(w_embed1 , c(max_words, ndim2)))


#wgt = fread("../input/embeddings/glove.840B.300d/glove.840B.300d.txt", data.table = FALSE,skip=1, verbose= TRUE)
wgt = fread("../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec", data.table = FALSE,skip=1, verbose= TRUE)

colnames(wgt)[1] <- "word"

wgt = wgt %>%
  mutate(word=gsub("[[:punct:]]"," ", rm_white(word) ))


dic_words = wgt$word
dic = data.frame(word=as.character(names(wordindex)), key = wordindex,row.names = NULL) %>%
  arrange(key) %>% 
  .[1:max_words,]
dic$word <- as.character(dic$word)

w_embed2 = dic %>% 
  left_join(wgt)
rm(wgt,dic); invisible(gc(reset=T))

J = ncol(w_embed2)
ndim2 = J-2
w_embed2 = w_embed2 [1:(max_words-1),3:J] %>%
  mutate_all(as.numeric) %>%
  mutate_all(round,6) %>%
  mutate_all(funs(replace(., is.na(.), 0))) 

colnames(w_embed2) = paste0("V",1:ndim2)
w_embed2 = rbind(rep(0, ndim2), w_embed2) %>%
  as.matrix()

w_embed2 = list(array(w_embed2 , c(max_words, ndim2)))



  #Model:
  inp1 = layer_input(shape = list(maxl))
  #inp2 = layer_input(shape = list(maxl))
  inp2 = layer_input(shape = list(ncol)) 
  emm1 = inp1 %>%
    layer_embedding(input_dim = max_words, output_dim = emb_dim, input_length = maxl, weights = w_embed,trainable=FALSE) 
  emm2 = inp1 %>%
    layer_embedding(input_dim = max_words, output_dim = emb_dim, input_length = maxl, weights = w_embed1,trainable=FALSE) 
  emm3 = inp1 %>%
    layer_embedding(input_dim = max_words, output_dim = emb_dim, input_length = maxl, weights = w_embed2,trainable=FALSE) 
 
 # Convolution
kernel_size = 5
filters = 64
pool_size = 4
  
 newinp =  layer_concatenate(list(emm1, emm2,emm3))
  
 model1 <- newinp %>% layer_spatial_dropout_1d(rate=0.1) %>%
  layer_conv_1d(filters, kernel_size, padding = "valid",activation = "relu",strides = 1)%>%
    bidirectional(layer_cudnn_lstm(units = 128, return_sequences = TRUE))%>%
    bidirectional(layer_cudnn_gru(units = 128, return_sequences = TRUE))%>%
                     layer_flatten() %>% 
      layer_dense(units = 280, activation = "relu") %>%
   layer_batch_normalization() %>%
   layer_dropout(rate=0.1) %>% 
   layer_dense(units = 16, activation = "relu") 

emm2 = inp2 %>%
    layer_dense(units = 1, activation = "sigmoid") %>%
    layer_dropout(rate=0.1) %>%
    layer_dense(units = 16, activation = "relu")
  
 
  outp = layer_concatenate(list(emm2, model1)) %>%
    layer_dense(units = 16, activation = "relu")%>%
    layer_dense(units = 1, activation = "sigmoid")
   
 model = keras_model(list(inp1,inp2), outp)
 
 summary(model)
 
  
  early_stopping <- callback_early_stopping(monitor = "val_loss",patience = 1,
                                            verbose = 1, mode = c( "min"))
  ###Loss & Metric Function
  
  
  F1Score <- R6::R6Class("F1Score",
                         inherit = KerasCallback,
                         
                         public = list(
                           
                           val = NA,
                           interval = NA,
                           
                           initialize = function(val, interval = 1) {
                             self$val <- val
                             self$interval <- interval
                           },
                           
                           on_epoch_end = function(epoch, logs) {
                             if (epoch %% self$interval == 0) {
                               y_pred <- round(self$model$predict(self$val[[1]]))
                               score <- ModelMetrics::f1Score(self$val[[2]], y_pred)
                               cat("F1 score on epoch", epoch+1, ":", score, "\n")
                             }
                           }
                         ))
  
  
  f1_loss <- function(y_true, y_pred) {
    
    tp <- k_sum(k_cast(y_true * y_pred, "float"), axis = 1)
    tn <- k_sum(k_cast((1 - y_true) * (1 - y_pred), "float"), axis = 1)
    
    fp <- k_sum(k_cast((1 - y_true) * y_pred, "float"), axis = 1)
    fn <- k_sum(k_cast(y_true * (1 - y_pred), "float"), axis = 1)
    
    p <- tp / (tp + fp + k_epsilon())
    r <- tp / (tp + fn + k_epsilon())
    
    f1 <- 2 * p * r / (p + r + k_epsilon())
    f1 <- tf$where(tf$is_nan(f1), tf$zeros_like(f1), f1)
    
    1 - k_mean(f1)
  }
  
  ###
  
  model %>% compile(
    optimizer = "adam",
    loss = f1_loss, metrics = "binary_accuracy"
  )
  
  early_stopping <- callback_early_stopping(patience = 3)
  
  #f1_score <- F1Score$new(list(x_val,val, y_val), 1)  
  check_point <- callback_model_checkpoint("model.h5", save_best_only = TRUE, verbose = 1, mode = "auto")
  
  # Fit with early stopping
  history = model %>% keras::fit(
    list(x_train1,tr), y_train,
    epochs = 20,
    batch_size = b_size,
    validation_data = list(list(x_val1,val),y_val),
    verbose=2,
    shuffle=TRUE,view_metrics = TRUE,
    callbacks = list(early_stopping, check_point))


pred <- data.frame(prediction=predict(model, list(x_val1,val)),truth=y_val)
sink(paste0("NN",".txt"))
#Threshold search
best.f1 <- best.thresh <- 0
for(thresh in seq(0.1,0.6,0.01)){
  preds.thresh <- ifelse(pred$prediction >= thresh, 1, 0)
  y_val <- pred$truth
  Precision <- caret::precision(data = factor(preds.thresh), reference = factor(y_val), relevant = "1")
  Recall <- caret::recall(data = factor(preds.thresh), reference = factor(y_val), relevant = "1")
  f1.thresh <- 2 * (Precision * Recall)/(Precision + Recall)
  cat("Thresh = ", thresh, "     ","F1 = ", f1.thresh,"\n")
  best.thresh <- ifelse(f1.thresh >= best.f1, thresh,best.thresh)
  best.f1 <- ifelse(f1.thresh >= best.f1, f1.thresh,best.f1)
}
sink()

###Prediction
print(history)
plot(history)
pred_nn <- ifelse(predict(model, list(x_test1,te))<best.thresh,0,1)
head(pred_nn,10)
read_csv("../input/sample_submission.csv")  %>%  
  mutate(prediction = pred_nn) %>%
  write_csv("submission.csv")