###Thanks to GolemKing for his Threshhold search function

library(devtools)
library(textfeatures)
library(rlang)
library(dplyr)
library(wordcloud)
library(ggraph)
library(igraph)
library(Rmisc)
library(SnowballC)
library(topicmodels)
library(glue)
library(scales)
library(tidytext)
library(text2vec)
library(stopwords)
library(Matrix)
library(tokenizers)
library(knitr)
library(keras)
library(tensorflow)
library(magrittr)
library(tidyverse)
library(rJava)
library(NLP)
library(openNLP)
library(RWeka)
library(magrittr)
library(tm)
library(caret)
library(stringi)
library(stringr)
library(topicmodels)
library(RTextTools)
library(readr)
library(data.table)
library(openNLPmodels.en)
library(NLP)
library(magrittr)
library(roperators)
getwd()
tr <- read_csv("../input/train.csv")
te <- read_csv("../input/test.csv")
subm <- read_csv("../input/sample_submission.csv")
library(caret)
y <- tr$target
tri <- caret::createDataPartition(y, p = 0.95, list = F) %>% c()
val <- tr[-tri,]
tr <- tr[tri,]
valy <- y[-tri]
y <- y[tri]
 tri <- 1:nrow(tr)
trid <- tr[,1]
trid <- data.frame(trid,stringasfactors=F)
#trid$trid <- as.character(trid$trid)
colnames(trid) <- c("qid")
valid <- val[,1]
teid <- te[,1]
tr_te <- tr[,-3] %>% rbind(te) %>% rbind(val[,-3])
rm(tr,te,val); gc()

###
past_simple = c("arose", "awoke"," was","were","bore","beat","became","began","bent","bet","bound","bit","bled","blew","broke","bred","brought","broadcast","built",
"burnt","burned","burst","bought","could","caught","chose","clung","came","cost","crept","cut","dealt","dug","did","drew","dreamt","dreamed","drank","drove","ate","fell",
"fed","felt","fought","found","flew","forbade","forgot","forgave","froze","got","gave","went","ground","grew","hung","had","heard","hid","hit","held","hurt","kept","knelt","knew",
"laid","led","leant","leaned","learnt","learned","left","lent","lay","lied","lit","lighted","lost","made","might","meant","met","mowed","had to","overtook","paid","put","read",
"rode","rang","rose","ran","sawed","said","saw","sold","sent","set","sewed","shook","should","shed","shone","shot","showed","shrank","shut","sang","sank",
"sat","slept","slid","smelt","sowed","spoke","spelt","spelled","spent","spilt","spilled","spat","spread","stood","stole","stuck","stung","stank","struck","swore","swept","swelled",
"swam","swung","took","taught","tore","told","thought","threw","understood","woke","wore","wept","would","won","wound","wrote")
present_tense <- c("arise","awake","be","bear","beat","become","begin","bend","bet","bind","bite","bleed","blow","break","breed","bring","broadcast","build","burn","burst","buy","can","catch","choose","cling","come","cost","creep","cut","deal","dig","do","draw","dream","drink","drive","eat","fall","feed","feel","fight","find","fly","forbid","forget","forgive","freeze","get","give","go","grind","grow","hang","have","hear","hide","hit","hold","hurt","keep","kneel","know","lay","lead","lean","learn","leave","lent","lie","light","lose","make","may","mean","meet","mow","must","overtake","pay","put","read","ride","ring","rise","run","saw","say","see","sell","send","set","sew","shake","shall","shed","shine","shoot","show","shrink","shut","sing","sink","sit","sleep","slide","smell","sow","speak","spell","spend","spill","spit","spread","stand","steal","stick","sting","stink","strike","swear","sweep","swell","swim","swing","take","teach","tear","tell","think","throw","understand","wake","wear","weep","will","win","wind","write")
past_participate = c("arisen","awoken","been","born","borne","beaten","become","begun","bent","bet","bound","bitten","bled","blown","broken","bred","brought","broadcast","built","burnt","burned","burst","bought","been able","caught","chosen","clung","come","cost","crept","cut","dealt","dug","done","drawn","dreamt","dreamed","drunk","driven","eaten","fallen","fed","felt","fought","found","flown","forbidden","forgotten","forgiven","frozen","got","given","gone","ground","grown","hung","had","heard","hidden","hit","held","hurt","kept","knelt","known","laid","led","leant","leaned","learnt","learned","left","lent","lain","lied","lit","lighted","lost","made","meant","met","mown","mowed","overtaken","paid","put","read","ridden","rung","risen","run","sawn","sawed","said","seen","sold","sent","set","sewn","sewed","shaken","shed","shone","shot","shown","shrunk","shut","sung","sunk","sat","slept","slid","smelt","sown","sowed","spoken","spelt","spelled","spent","spilt","spilled","spat","spread","stood","stolen","stuck","stung","stunk","struck","sworn","swept","swollen","swelled","swum","swung","taken","taught","torn","told","thought","thrown","understood","woken","worn","wept","won","wound","written")
frequency_adverb <- c("always","usually","often","normally","occasionally ","sometimes","seldom","never","hardly","ever","constantly","continually","frequently","infrequently","intermittently","periodically","rarely","regularly","generally","now and then","almost never","eventually","quarterly","weekly","later","then")
time_adverb <- c("now","then","today","tomorrow","tonight","yesterday","annually","daily","fortnightly","hourly","monthly","nightly","quarterly","weekly","yearly","always","constantly","ever","frequently","generally","infrequently","never","normally","occasionally","often","rarely","regularly","seldom","sometimes","regularly","usually","already","before","early","earlier","eventually","finally","first","formerly","just","last","late","later","lately","next","previously","recently","since","soon","still","yet")
place_adverb <- c("about","above","abroad","anywhere","away","back","backwards","behind","below","down","downstairs","east","west","north","south","elsewhere","far","here","in","indoors","inside","near","nearby","off","on","out","outside","over","there","towards","under","up","upstairs","where","everywhere","somewhere","nowhere","somewhere","eastward","westward","eastwards","westwards","forwards","homewards","upwards")
manner_adverb <- c("accidentally","angrily","anxiously","awkwardly","badly","beautifully","blindly","boldly","bravely","brightly","busily","calmly","carefully","carelessly","cautiously","cheerfully","clearly","closely","correctly","courageously","cruelly","daringly","deliberately","doubtfully","eagerly","easily","elegantly","enormously","enthusiastically","equally","eventually","exactly","faithfully","fast","fatally","fiercely","fondly","foolishly","fortunately","frankly","frantically","generously","gently","gladly","gracefully","greedily","happily","hard","hastily","healthily","honestly","hungrily","hurriedly","inadequately","ingeniously","innocently","inquisitively","irritably","joyously","justly","kindly","lazily","loosely","loudly","madly","mortally","mysteriously","neatly","nervously","noisily","obediently","openly","painfully","patiently","perfectly","politely","poorly","powerfully","promptly","punctually","quickly","quietly","rapidly","rarely","really","recklessly","regularly","reluctantly","repeatedly","rightfully","roughly","rudely","sadly","safely","selfishly","sensibly","seriously","sharply","shyly","silently","sleepily","slowly","smoothly","so","softly","solemnly","speedily","stealthily","sternly","straight","stupidly","successfully","suddenly","suspiciously","swiftly","tenderly","tensely","thoughtfully","tightly","truthfully","unexpectedly","victoriously","violently","vivaciously","warmly","weakly","wearily","well","wildly","wisely")
degree_adverb <- c("extremely","terribly","amazingly","wonderfully","insanely","especially","particularly","uncommonly","unusually","remarkably","quite","pretty","rather","fairly","not especially","not particularly","never","rarely","not only","scarcely","seldom","very","too","enough","just","almost")
Noun_of_action_suffix <- c("ise","ize","ism","ist")
possessive_noun <- c("s","s")
Derivational_suffix <- c("ness","less","ity","ion", "ify")
firstperson_pronoun <- c(" i ", " me " ,"we", "us")
secondperson_pronoun <- c("you")
thirdperson_pronoun <- c("she","her","he","him","it","they","them")
relativepronoun <- c("that","which","who","whom","whose","whichever","whoever","whatever")
demonstrative_pronoun <- c("this","that","these","those")
indefinite_pronoun <- c("anybody","anyone","anything","each","either","everybody","everyone","everything","neither","nobody","no one","nothing","one","somebody","someone","something","both","few","many","several","all","any","most","none","some")
reflexive_pronoun <- c("self","selves")
interrogative_pronoun <- c("what","who","where","when","whose","which","whom")
possessive_pronoun <- c("my","mine","your","yours","his","her","its","our","their","ours","theirs","his","hers")
subject_pronoun <- c(" i ", "you","she","he","it","we","you","they")
object_pronoun <- c("you","me","her","him","it","us","them")
nominalization <- c("tion", "ment", "ness", "ity")
gerunds <- c("ing")
cojunct <- c("after","although","as","as if","as long as","as much as","as soon as","as though","because","before","by the time","even if","even though","I if","in order that","in case","lest","once","only if","provided that","unless","until","when","whenever","where","wherever","while","and","or","either","neither","nor","not only","but also","whether")
hedges <- c("usually","generally","relatively","almost","at least","nearly","roughly","typically","potentially","ultimately","around","approximately","seems to","for the most part","more or less","on average","nearly","in the neighborhood of","upwards of")
inflators <- c("very","highly","extremely","literally","truly","really","totally","greatly","key","immediately","suddenly","precisely","absolutely","intrinsically","very important to note","specific key concept")
discourse_marker <- c("anyway","like","right","you know","fine","now","so","I mean","good","oh","well","as I say","great","okay","mind you","for a start","firstly","in addition","moreover","on the other hand","secondly","in conclusion","on the one hand","to begin with","thirdly","in sum ")
modal_possiblity <- c("can", "may", "might", "could")
modal_necessity <- c("ought","should","must")
modal_predictive <- c("will","would","shall")
verb_seem <- c("seem")
public_verb <- c("affirm", "announce", "boast", "confirm", "declare","state","confirm","report","sai","interview","talk","mention")
private_verb <- c("think","deduce","assume", "believe", "doubt", "know","fear","guess","like","remember","wish","hope","forget","underst","wonder","feel","love","hate")
power_verbs <- c("abolish","accelerate","achieve","act","adopt","align","anticipate","apply","assess","avoid","boost","break","bridge","build","burn","capture","change","choose","clarify","clobber","confront","connect","conquer","convert","create","decide","define","defuse","deliver","deploy","design","develop","diagnose","discover","drive","eliminate","ensure","establish","evaluate","exploit","explore","filter","finalize","find","focus","foresee","gain","gather","generate","grasp","identify","ignite","implement","improve","increase","innovate","inspire","intensify","lead","learn","leverage","manage","master","maximize","measure","mobilize","motivate","overcome","penetrate","persuade","plan","position","prepare","prevent","profit","raise","reconsider","reduce","refresh","replace","resist","respond","retain","save","scan","shatter","shadeoff","sidestep","simplify","slash","solve","stimulate","stop","stretch","succeed","supplement","take","transfer","transform","understand","unleash","unravel","use","win")
General_swear_words <- c("twat","tits","son of a bitch","sod off","snatch","shit","pussy","punani","prick","piss","munter","mofo","fuck","minge","knob","damn","ginger","gash","flaps","eff","feck","dick","cunt","crap","cow","clunge","bull","bugger","bollocks","bitch","beaver","ass","arse","bastard","balls")
Sexual_references <- c("bonk","bukkake","cocksucker","dildo","ho","slut","jizz","nonce","prickteaser","rape","shag","skank","slag","skank","slapper","tart","wanker","whore")
Age_dsicrim <- c("coffin dodger","fop","old bag")
Religion_racist <- c("fenian","kafir","kufaar","kike","papist","prod","taig","yid","muslim","arab","dalit","caste","sect","jihad")
orientation_identity <- c("batty boy","bender","bum boy","bumclat","bummer","chi-chi","dyke","fag","fudge","gay","homo","lesbo","lesb","lezza","muff","pansy","poof","queer","rugmuncher","tranny")
mental_physicalhealth <- c("cretin","cripple","div","loony","mental","midget","mong","nutter","psycho","retard","schizo","spakka","spaz","spastic")
Google_banned_badwords <- c("4r5e","5h1t","5hit","a55","anal","anus","ar5e","arrse","arse","ass","ass-fucker","asses","assfucker","assfukka","asshole","assholes","asswhole","a_s_s","b!tch","b00bs","b17ch","b1tch","ballbag","balls","ballsack","bastard","beastial","beastiality","bellend","bestial","bestiality","bi+ch","biatch","bitch","bitcher","bitchers","bitches","bitchin","bitching","bloody","blow job","blowjob","blowjobs","boiolas","bollock","bollok","boner","boob","boobs","booobs","boooobs","booooobs","booooooobs","breasts","buceta","bugger","bum","bunny fucker","butt","butthole","buttmuch","buttplug","c0ck","c0cksucker","carpet muncher","cawk","chink","cipa","cl1t","clit","clitoris","clits","cnut","cock","cock-sucker","cockface","cockhead","cockmunch","cockmuncher","cocks","cocksuck ","cocksucked ","cocksucker","cocksucking","cocksucks ","cocksuka","cocksukka","cok","cokmuncher","coksucka","coon","cox","crap","cum","cummer","cumming","cums","cumshot","cunilingus","cunillingus","cunnilingus","cunt","cuntlick ","cuntlicker ","cuntlicking ","cunts","cyalis","cyberfuc","cyberfuck ","cyberfucked ","cyberfucker","cyberfuckers","cyberfucking ","d1ck","damn","dick","dickhead","dildo","dildos","dink","dinks","dirsa","dlck","dog-fucker","doggin","dogging","donkeyribber","doosh","duche","dyke","ejaculate","ejaculated","ejaculates ","ejaculating ","ejaculatings","ejaculation","ejakulate","f u c k","f u c k e r","f4nny","fag","fagging","faggitt","faggot","faggs","fagot","fagots","fags","fanny","fannyflaps","fannyfucker","fanyy","fatass","fcuk","fcuker","fcuking","feck","fecker","felching","fellate","fellatio","fingerfuck ","fingerfucked ","fingerfucker ","fingerfuckers","fingerfucking ","fingerfucks ","fistfuck","fistfucked ","fistfucker ","fistfuckers ","fistfucking ","fistfuckings ","fistfucks ","flange","fook","fooker","fuck","fucka","fucked","fucker","fuckers","fuckhead","fuckheads","fuckin","fucking","fuckings","fuckingshitmotherfucker","fuckme ","fucks","fuckwhit","fuckwit","fudge packer","fudgepacker","fuk","fuker","fukker","fukkin","fuks","fukwhit","fukwit","fux","fux0r","f_u_c_k","gangbang","gangbanged ","gangbangs ","gaylord","gaysex","goatse","God","god-dam","god-damned","goddamn","goddamned","hardcoresex ","hell","heshe","hoar","hoare","hoer","homo","hore","horniest","horny","hotsex","jack-off ","jackoff","jap","jerk-off ","jism","jiz ","jizm ","jizz","kawk","knob","knobead","knobed","knobend","knobhead","knobjocky","knobjokey","kock","kondum","kondums","kum","kummer","kumming","kums","kunilingus","l3i+ch","l3itch","labia","lmfao","lust","lusting","m0f0","m0fo","m45terbate","ma5terb8","ma5terbate","masochist","master-bate","masterb8","masterbat*","masterbat3","masterbate","masterbation","masterbations","masturbate","mo-fo","mof0","mofo","mothafuck","mothafucka","mothafuckas","mothafuckaz","mothafucked ","mothafucker","mothafuckers","mothafuckin","mothafucking ","mothafuckings","mothafucks","mother fucker","motherfuck","motherfucked","motherfucker","motherfuckers","motherfuckin","motherfucking","motherfuckings","motherfuckka","motherfucks","muff","mutha","muthafecker","muthafuckker","muther","mutherfucker","n1gga","n1gger","nazi","nigg3r","nigg4h","nigga","niggah","niggas","niggaz","nigger",
                            "niggers ","nob","nob jokey","nobhead","nobjocky","nobjokey","numbnuts","nutsack","orgasim ","orgasims ","orgasm","orgasms ","p0rn","pawn","pecker","penis","penisfucker","phonesex","phuck","phuk","phuked","phuking","phukked","phukking","phuks","phuq","pigfucker","pimpis","piss","pissed","pisser","pissers","pisses ","pissflaps","pissin ","pissing","pissoff ","poop","porn","porno","pornography","pornos","prick","pricks ","pron","pube","pusse","pussi","pussies","pussy","pussys ","rectum","retard","rimjaw","rimming","s hit","s.o.b.","sadist","schlong","screwing","scroat","scrote","scrotum","semen","sex","sh!+","sh!t","sh1t","shag","shagger","shaggin","shagging","shemale","shi+","shit","shitdick","shite","shited","shitey","shitfuck","shitfull","shithead","shiting","shitings","shits","shitted","shitter","shitters ","shitting","shittings","shitty ","skank","slut","sluts","smegma","smut","snatch","son-of-a-bitch","spac","spunk","s_h_i_t","t1tt1e5","t1tties","teets","teez","testical","testicle","tit","titfuck","tits","titt","tittie5","tittiefucker","titties","tittyfuck","tittywank","titwank","tosser","turd","tw4t","twat","twathead","twatty","twunt","twunter","v14gra","v1gra","vagina","viagra","vulva","w00se","wang","wank","wanker","wanky","whoar","whore","willies","willy","xrated","xxx")
bad_words <- c("abbo","abo","abortion","abuse","addict","addicts","adult","africa","african",
               "alla","allah","alligatorbait","amateur","american","anal","analannie","analsex",
               "angie","angry","anus","arab","arabs","areola","argie","aroused","arse","arsehole",
               "asian","ass","assassin","assassinate","assassination","assault","assbagger","assblaster",
               "assclown","asscowboy", "asses","assfuck","assfucker","asshat","asshole","assholes","asshore",
               "assjockey","asskiss","asskisser","assklown","asslick","asslicker","asslover","assman","assmonkey",
               "assmunch","assmuncher","asspacker","asspirate","asspuppies","assranger","asswhore","asswipe",
               "athletesfoot","attack","australian","babe","babies","backdoor","backdoorman","backseat",
               "badfuck","balllicker","balls","ballsack","banging","baptist","barelylegal","barf","barface",
               "barfface","bast","bastard","bazongas","bazooms","beaner","beast","beastality","beastial",
               "beastiality","beatoff","beat-off","beatyourmeat","beaver","bestial","bestiality","bi","biatch",
               "bible","bicurious","bigass","bigbastard","bigbutt","bigger","bisexual","bi-sexual","bitch",
               "bitcher","bitches","bitchez","bitchin","bitching","bitchslap","bitchy","biteme","black",
               "blackman","blackout","blacks","blind","blow","blowjob","boang","bogan","bohunk","bollick",
               "bollock","bomb","bombers","bombing","bombs","bomd","bondage","boner","bong","boob","boobies",
               "boobs","booby","boody","boom","boong","boonga","boonie","booty","bootycall","bountybar","bra",
               "brea5t","breast","breastjob","breastlover","breastman","brothel","bugger","buggered","buggery",
               "bullcrap","bulldike","bulldyke","bullshit","bumblefuck","bumfuck","bunga","bunghole","buried",
               "burn","butchbabes","butchdike","butchdyke","butt","buttbang","butt-bang","buttface","buttfuck",
               "butt-fuck","buttfucker","butt-fucker","buttfuckers","butt-fuckers","butthead","buttman",
               "buttmunch","buttmuncher","buttpirate","buttplug","buttstain","byatch","cacker","cameljockey",
               "cameltoe","canadian","cancer","carpetmuncher","carruth","catholic","catholics","cemetery","chav",
               "cherrypopper","chickslick","chin","chinaman","chinamen","chinese","chink","chinky","choad",
               "chode","christ","christian","church","cigarette","cigs","clamdigger","clamdiver","clit",
               "clitoris","clogwog","cocaine","cock","cockblock","cockblocker","cockcowboy","cockfight",
               "cockhead","cockknob","cocklicker","cocklover","cocknob","cockqueen","cockrider","cocksman",
               "cocksmith","cocksmoker","cocksucer","cocksuck ","cocksucked ","cocksucker","cocksucking",
               "cocktail","cocktease","cocky","cohee","coitus","color","colored","coloured","commie","communist",
               "condom","conservative","conspiracy","coolie","cooly","coon","coondog","copulate","cornhole",
               "corruption","cra5h","crabs","crack","crackpipe","crackwhore","crack-whore","crap","crapola",
               "crapper","crappy","crash","creamy","crime","crimes","criminal","criminals","crotch",
               "crotchjockey","crotchmonkey","crotchrot","cum","cumbubble","cumfest","cumjockey","cumm",
               "cummer","cumming","cumquat","cumqueen","cumshot","cunilingus","cunillingus","cunn","cunnilingus",
               "cunntt","cunt","cunteyed","cuntfuck","cuntfucker","cuntlick","cuntlicker","cuntlicking",
                "cuntsucker","cybersex","cyberslimer","dago","dahmer","dammit","damn","damnation","damnit",
                "darkie","darky","datnigga","dead","deapthroat","death","deepthroat","defecate","dego","demon",
"deposit","desire","destroy","deth","devil","devilworshipper","dick","dickbrain","dickforbrains","dickhead",
"dickless","dicklick","dicklicker","dickman","dickwad","dickweed","diddle","die","died","dies",
"dike","dildo","dingleberry","dink","dipshit","dipstick","dirty","disease","diseases","disturbed",
"dive","dix","dixiedike","dixiedyke","doggiestyle","doggystyle","dong","doodoo","doo-doo","doom",
"dope","dragqueen","dragqween","dripdick","drug","drunk","drunken","dumb","dumbass","dumbbitch",
"dumbfuck","dyefly","dyke","easyslut","eatballs","eatme","eatpussy","ecstacy","ejaculate",
"ejaculated","ejaculating","ejaculation","enema","enemy","erect","erection","ero","escort",
"ethiopian","ethnic","european","evl","excrement","execute","executed","execution","executioner",
"explosion","facefucker","faeces","fag","fagging","faggot","fagot","failed","failure","fairies",
"fairy","faith","fannyfucker","fart","farted","farting ","farty ","fastfuck","fat","fatah","fatass",
"fatfuck","fatfucker","fatso","fckcum","fear","feces","felatio ","felch","felcher","felching","fellatio",
"feltch","feltcher","feltching","fetish","fight","filipina","filipino","fingerfood","fingerfuck ","fingerfucked",
"fingerfucker ","fingerfuckers","fingerfucking ","fire","firing","fister","fistfuck","fistfucked","fistfucker","fistfucking",
"fisting","flange","flasher","flatulence","floo","flydie","flydye","fok","fondle","footaction","footfuck","footfucker","footlicker",
"footstar","fore","foreskin","forni","fornicate","foursome","fourtwenty","fraud","freakfuck","freakyfucker","freefuck","fu","fubar",
"fuc","fucck","fuck","fucka","fuckable","fuckbag","fuckbuddy","fucked","fuckedup","fucker","fuckers","fuckface","fuckfest","fuckfreak",
"fuckfriend","fuckhead","fuckher","fuckin","fuckina","fucking","fuckingbitch","fuckinnuts","fuckinright","fuckit","fuckknob","fuckme ",
"fuckmehard","fuckmonkey","fuckoff","fuckpig","fucks","fucktard","fuckwhore","fuckyou","fudgepacker","fugly","fuk","fuks","funeral",
"funfuck","fungus","fuuck","gangbang","gangbanged ","gangbanger","gangsta","gatorbait","gay","gaymuthafuckinwhore","gaysex ","geez",
"geezer","geni","genital","german","getiton","gin","ginzo","gipp","girls","givehead","glazeddonut","gob","god","godammit","goddamit",
"goddammit","goddamn","goddamned","goddamnes","goddamnit","goddamnmuthafucker","goldenshower","gonorrehea","gonzagas","gook","gotohell",
"goy","goyim","greaseball","gringo","groe","gross","grostulation","gubba","gummer","gun","gyp","gypo","gypp","gyppie","gyppo","gyppy",
"hamas","handjob","hapa","harder","hardon","harem","headfuck","headlights","hebe","heeb","hell","henhouse","heroin","herpes","heterosexual",
"hijack","hijacker","hijacking","hillbillies","hindoo","hiscock","hitler","hitlerism","hitlerist","hiv","ho","hobo","hodgie","hoes","hole",
"holestuffer","homicide","homo","homobangers","homosexual","honger","honk","honkers","honkey","honky","hook","hooker","hookers","hooters",
"hore","hork","horn","horney","horniest","horny","horseshit","hosejob","hoser","hostage","hotdamn","hotpussy","hottotrot","hummer","husky",
"hussy","hustler","hymen","hymie","iblowu","idiot","ikey","illegal","incest","insest","intercourse","interracial","intheass","inthebuff",
"israel","israeli","italiano","itch","jackass","jackoff","jackshit","jacktheripper","jade","jap","japanese","japcrap","jebus","jeez",
"jerkoff","jesus","jesuschrist","jew","jewish","jiga","jigaboo","jigg","jigga","jiggabo","jigger ","jiggy","jihad","jijjiboo",
"jimfish","jism","jiz ","jizim","jizjuice","jizm ","jizz","jizzim","jizzum","joint","juggalo","jugs","junglebunny","kaffer",
"kaffir","kaffre","kafir","kanake","kid","kigger","kike","kill","killed","killer","killing","kills","kink","kinky","kissass",
"kkk","knife","knockers","kock","kondum","koon","kotex","krap","krappy","kraut","kum","kumbubble","kumbullbe","kummer","kumming",
"kumquat","kums","kunilingus","kunnilingus","kunt","ky","kyke","lactate","laid","lapdance","latin","lesbain","lesbayn","lesbian",
"lesbin","lesbo","lez","lezbe","lezbefriends","lezbo","lezz","lezzo","liberal","libido","licker","lickme","lies","limey","limpdick",
"limy","lingerie","liquor","livesex","loadedgun","lolita","looser","loser","lotion","lovebone","lovegoo","lovegun","lovejuice","lovemuscle",
"lovepistol","loverocket","lowlife","lsd","lubejob","lucifer","luckycammeltoe","lugan","lynch","macaca","mad","mafia","magicwand","mams","manhater",
"manpaste","marijuana","mastabate","mastabater","masterbate","masterblaster","mastrabator","masturbate","masturbating","mattressprincess","meatbeatter",
"meatrack","meth","mexican","mgger","mggor","mickeyfinn","mideast","milf","minority","mockey","mockie","mocky","mofo","moky","moles","molest","molestation",
"molester","molestor","moneyshot","mooncricket","mormon","moron","moslem","mosshead","mothafuck","mothafucka","mothafuckaz","mothafucked ","mothafucker","mothafuckin",
"mothafucking ","mothafuckings","motherfuck","motherfucked","motherfucker","motherfuckin","motherfucking","motherfuckings","motherlovebone","muff","muffdive","muffdiver",
"muffindiver","mufflikcer","mulatto","muncher","munt","murder","murderer","muslim","naked","narcotic","nasty","nastybitch","nastyho","nastyslut","nastywhore","nazi","necro",
"negro","negroes","negroid","negro","nig","niger","nigerian","nigerians","nigg","nigga","niggah","niggaracci","niggard","niggarded","niggarding","niggardliness","niggardlinesss",
"niggardly","niggards","niggard","niggaz","nigger","niggerhead","niggerhole","niggers","niggers","niggle","niggled","niggles","niggling","nigglings","niggor","niggur","niglet",
"nignog","nigr","nigra","nigre","nip","nipple","nipplering","nittit","nlgger","nlggor","nofuckingway","nook","nookey","nookie","noonan","nooner","nude","nudger","nuke","nutfucker",
"nymph","ontherag","oral","orga","orgasim ","orgasm","orgies","orgy","osama","paki","palesimian",
"palestinian","pansies","pansy","panti","panties","payo","pearlnecklace","peck","pecker","peckerwood",
"pee","peehole","pee-pee","peepshow","peepshpw","pendy","penetration","peni5","penile","penis","penises",
"penthouse","period","perv","phonesex","phuk","phuked","phuking","phukked","phukking","phungky","phuq",
"pi55","picaninny","piccaninny","pickaninny","piker","pikey","piky","pimp","pimped","pimper","pimpjuic",
"pimpjuice","pimpsimp","pindick","piss","pissed","pisser","pisses ","pisshead","pissin ","pissing",
"pissoff ","pistol","pixie","pixy","playboy","playgirl","pocha","pocho","pocketpool","pohm","polack",
"pom","pommie","pommy","poo","poon","poontang","poop","pooper","pooperscooper","pooping","poorwhitetrash",
"popimp","porchmonkey","porn","pornflick","pornking","porno","pornography","pornprincess","pot","poverty",
"premature","pric","prick","prickhead","primetime","propaganda","pros","prostitute","protestant","pu55i",
"pu55y","pube","pubic","pubiclice","pud","pudboy","pudd","puddboy","puke","puntang","purinapricness","puss",
"pussie","pussies","pussy","pussycat","pussyeater","pussyfucker","pussylicker","pussylips","pussylover",
"pussypounder","pusy","quashie","queef","queer","quickie","quim","ra8s","rabbi","racial","racist","radical",
"radicals","raghead","randy","rape","raped","raper","rapist","rearend","rearentry","rectum","redlight",
"redneck","reefer","reestie","refugee","reject","remains","rentafuck","republican","rere","retard",
"retarded","ribbed","rigger","rimjob","rimming","roach","robber","roundeye","rump","russki","russkie",
"sadis","sadom","samckdaddy","sandm","sandnigger","satan","scag","scallywag","scat","schlong","screw",
"screwyou","scrotum","scum","semen","seppo","servant","sex","sexed","sexfarm","sexhound","sexhouse",
"sexing","sexkitten","sexpot","sexslave","sextogo","sextoy","sextoys","sexual","sexually","sexwhore",
"sexy","sexymoma","sexy-slim","shag","shaggin","shagging","shat","shav","shawtypimp","sheeney","shhit",
"shinola","shit","shitcan","shitdick","shite","shiteater","shited","shitface","shitfaced","shitfit",
"shitforbrains","shitfuck","shitfucker","shitfull","shithapens","shithappens","shithead","shithouse",
"shiting","shitlist","shitola","shitoutofluck","shits","shitstain","shitted","shitter","shitting",
"shitty ","shoot","shooting","shortfuck","showtime","sick","sissy","sixsixsix","sixtynine","sixtyniner",
"skank","skankbitch","skankfuck","skankwhore","skanky","skankybitch","skankywhore","skinflute","skum",
"skumbag","slant","slanteye","slapper","slaughter","slav","slave","slavedriver","sleezebag",
"sleezeball","slideitin","slime","slimeball","slimebucket","slopehead","slopey","slopy","slut",
"sluts","slutt","slutting","slutty","slutwear","slutwhore","smack","smackthemonkey","smut",
"snatch","snatchpatch","snigger","sniggered","sniggering","sniggers","sniggers","sniper",
"snot","snowback","snownigger","sob","sodom","sodomise","sodomite","sodomize","sodomy",
"sonofabitch","sonofbitch","sooty","sos","soviet","spaghettibender","spaghettinigger",
"spank","spankthemonkey","sperm","spermacide","spermbag","spermhearder","spermherder",
"spic","spick","spig","spigotty","spik","spit","spitter","splittail","spooge","spreadeagle",
"spunk","spunky","squaw","stagg","stiffy","strapon","stringer","stripclub","stroke","stroking","stupid",
"stupidfuck","stupidfucker","suck","suckdick","sucker","suckme","suckmyass","suckmydick","suckmytit",
"suckoff","suicide","swallow","swallower","swalow","swastika","sweetness","syphilis","taboo","taff",
"tampon","tang","tantra","tarbaby","tard","teat","terror","terrorist","teste","testicle","testicles",
"thicklips","thirdeye","thirdleg","threesome","threeway","timbernigger","tinkle","tit","titbitnipply",
"titfuck","titfucker","titfuckin","titjob","titlicker","titlover","tits","tittie","titties","titty",
"tnt","toilet","tongethruster","tongue","tonguethrust","tonguetramp","tortur","torture","tosser",
"towelhead","trailertrash","tramp","trannie","tranny","transexual","transsexual","transvestite",
"triplex","trisexual","trojan","trots","tuckahoe","tunneloflove","turd","turnon","twat","twink",
"twinkie","twobitwhore","uck","uk","unfuckable","upskirt","uptheass","upthebutt","urinary","urinate",
"urine","usama","uterus","vagina","vaginal","vatican","vibr","vibrater","vibrator","vietcong","violence",
"virgin","virginbreaker","vomit","vulva","wab","wank","wanker","wanking","waysted","weapon","weenie","weewee",
"welcher","welfare","wetb","wetback","wetspot","whacker","whash","whigger","whiskey","whiskeydick","whiskydick",
"whit","whitenigger","whites","whitetrash","whitey","whiz","whop","whore","whorefucker","whorehouse","wigger","willie",
"williewanker","willy","wn","wog","womens","wop","wtf","wuss","wuzzie","xtc","xxx","yankee","yellowman","zigabo","zipperhead")
Race_ethnicity <- c("chink","choc ice","colored","coloured","coon","darky","dago","gippo","golliwog","jock","honky","hun","jap","kraut","nazi","negro","nigg","paki","pikey","polack","raghead","sambo","slope","spic","taff","wog","wop")
country_nationality <- c("China","Armenia","Argentina","Afghanistan","Albania","United Arab Emirates","Austria","Spain","France","United States","Germany","Brazil","Switzerland","Côte d'Ivoire","Chile","Romania","Portugal","Paraguay","Réunion",
                         "Poland","Uruguay","Uzbekistan","Venezuela, Bolivarian Rep. of","Viet Nam","Vanuatu","Afghan",
                         "Afghanistan","Afrikaans","Albania","Albanian","Algeria","Algerian","American","American Samoa",
                         "Amharic","Andorra","Angola","Anguilla","Antigua and Barbuda","Arabic","Arabic, Kurdish","Arabiv",
                         "Argentina","Argentine","Argentinian","Armenia","Aruba","Australia","Australian","Austria",
                         "Austrian","Azerbaijan","Bahamas","Bahrain","Bangladesh","Bangladeshi","Barbados","Batswana",
                         "Belarus","Belgian","Belgium","Belize","Bengali","Benin","Bermuda","Bhutan","Bolivia","Bolivian",
                         "Bosnia and Herzegovina","Botswana","Brazil","Brazilian","British","British Virgin Islands",
                         "Brunei Darussalam","Bulgaria","Bulgarian","Burkina Faso","Burundi","Cambodia","Cambodian",
                         "Cameroon","Cameroonian","Canada","Canadian","Cape Verde","Cayman Islands",
                         "Central African Republic","Chad","Chile","Chilean","China","Chinese","Colombia","Colombian",
                         "Comoros","Congo","Congo, Democratic Republic of the","Cook Islands","Costa Rica","Costa Rican",
                         "Côte d'Ivoire","Country","Croatia","Croatian","Cuba","Cuban","Cyprus","Czech","Czech Republic",
                         "Danish","Denmark","Djibouti","Dominica","Dominican","Dominican Republic","Dutch","Ecuador",
                         "Ecuadorian","Egypt","Egyptian","El Salvador","Emirati","English","English ",
                         "Equatorial Guinea","Eritrea","Estonia","Estonian","Ethiopia","Ethiopian",
                         "Faeroe Islands","Falkland Islands (Malvinas)","Fiji","Fijian","Finland","Finnish",
                         "Flemish","France","French","French Guiana","French Polynesia","Gabon","Gambia",
                         "Georgia","German","Germany","Ghana","Ghanaian","Gibraltar","Greece","Greek","Greenland",
                         "Grenada","Guadeloupe","Guam","Guatemala","Guatemalan","Guernsey","Guinea","Guinea-Bissau",
                         "Guyana","Haiti","Haitian","Hebrew","Hindi","Honduran","Honduras","Hong Kong","Hungarian",
                         "Hungary","Iceland","Icelandic","India","Indian","Indonesia","Indonesian","Iran","Iranian",
                         "Iraq","Iraqi","Ireland","Irish","Isle of Man","Israel","Israeli","Italian","Italy","Jamaica",
                         "Jamaican","Japan","Japanese","Jersey","Jordan","Jordanian","Kazakhstan","Kenya","Kenyan",
                         "Kiribati","Korea, Dem. People's Rep. of","Korea, Republic of","Korean","Kuwait","Kuwaiti",
                         "Kyrgyzstan","Lao","Lao People's Dem. Rep.","Laotian","Latvia","Latvian","Lebanese","Lebanon",
                         "Lesotho","Liberia","Libyan","Libyan Arab Jamahiriya","Liechtenstein","Lithuania","Lithuanian",
                         "Luxembourg","Macau, China","Macedonia, The former Yugoslav Rep. of","Madagascar",
                         "Malawi","Malay","Malaysia","Malaysian","Maldives","Mali","Malian","Malta","Maltese",
                         "Marshall Islands","Martinique","Mauritania","Mauritius","Mexican","Mexico","Moldova",
                         "Monaco","Mongolia","Mongolian","Montenegro","Montserrat","Moroccan","Morocco","Mozambican",
                         "Mozambique","Myanmar","Namibia","Namibian","Nauru","Nepal","Nepalese","Nepali","Netherlands",
                         "New Caledonia","New Zealand","Nicaragua","Nicaraguan","Niger","Nigeria","Nigerian","Niue",
                         "Norfolk Island","Northern Mariana Islands","Norway","Norwegian","Oman","Pakistan","Pakistani",
                         "Palau","Panama","Panamanian","Papua New Guinea","Paraguay","Paraguayan","Pashto","Persian",
                         "Peru","Peruvian","Philippine","Philippines","Poland","Polish","Portugal","Portuguese",
                         "Puerto Rico","Qatar","Réunion","Romania","Romanian","Russian","Russian Federation","Rwanda",
                         "Saint Helena","Saint Kitts and Nevis","Saint Lucia","Saint Pierre and Miquelon",
                         "Saint Vincent and the Grenadines","Salvadorian","Samoa","San Marino",
                         "Sao Tome and Principe","Saudi","Saudi Arabia","Scottish","Senegal",
                         "Senegalese","Serbia","Serbian","Seychelles","Sierra Leone","Singapore",
                         "Singaporean","Sinhala","Slovak","Slovakia","Slovenia","Solomon Islands",
                         "Somalia","Africa","South African","Sudan","Spain","Spanish","Sri Lanka",
                         "Sri Lankan","Sudan","Sudanese","Suriname","Swahili","Swaziland","Sweden",
                         "Swedish","Swiss","Switzerland","Syrian","Syrian Arab Republic","Tagalog",
                         "Taiwan, China","Taiwanese","Tajik","Tajikistan","Tajikistani","Tamil",
                         "Tanzania","Thai","Thailand","Timor-Leste","Togo","Tokelau","Tonga","Tongan",
                         "Trinidad and Tobago","Tunisia","Tunisian","Turkey","Turkish","Turkmenistan",
                         "Turks","Caicos","Tuvalu","Uganda","Ukraine","Ukrainian",
                         "United Arab Emirates","United Kingdom","United States",
                         "Urdu","Uruguay","Uruguayan","Uzbekistan","Vanuatu",
                         "Venezuela","Venezuelan","Viet Nam","Vietnamese",
                         "Virgin Islands","Wallis and Futuna Islands","Welsh",
                         "West Bank and Gaza Strip","Western Sahara","Yemen","Zambia","Zambian",
                         "Zimbabwe"," Setswana","Afghan","Albanian","Algerian","Argentine",
                         "Argentinian","Australian","Austrian","Bangladeshi","Belgian",
                         "Bolivian","Batswana","Brazilian","Bulgarian","Cambodian","Cameroonian","Canadian","Chilean",
                         "Chinese","Colombian","Costa Rican","Croatian","Cuban","Czech","Danish","Dominican","Ecuadorian",
                         "Egyptian","Salvadorian","English")
library(stringi)
library(stringr)
tr_te$country_nationality <- laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, country_nationality)))
tr_te$bad_words <- laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, bad_words)))
summary(tr_te$country_nationality)
tr_te$General_swear_words =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, General_swear_words)))
tr_te$Google_banned_badwords <- laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, Google_banned_badwords)))
tr_te$Sexual_references =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, Sexual_references)))
tr_te$Age_dsicrim =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, Age_dsicrim)))
tr_te$Religion_racist =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, Religion_racist)))
tr_te$orientation_identity =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, orientation_identity)))
tr_te$mental_physicalhealth =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, mental_physicalhealth)))
tr_te$Race_ethnicity =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, Race_ethnicity)))
tr_te$past_participate =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, past_participate)))
tr_te$past_simple =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, past_simple)))
tr_te$present_tense =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, present_tense)))
tr_te$frequency_adverb =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, frequency_adverb)))
tr_te$time_adverb =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, time_adverb)))
tr_te$place_adverb =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, place_adverb)))
tr_te$manner_adverb =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, manner_adverb)))
tr_te$degree_adverb =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, degree_adverb)))
tr_te$Noun_of_action_suffix =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, Noun_of_action_suffix)))
tr_te$possessive_noun =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, possessive_noun)))
tr_te$Derivational_suffix =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, Derivational_suffix)))
tr_te$nominalization =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, nominalization)))
tr_te$firstperson_pronoun =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, firstperson_pronoun)))
tr_te$secondperson_pronoun =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, secondperson_pronoun)))
tr_te$thirdperson_pronoun =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, thirdperson_pronoun)))
tr_te$relativepronoun =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, relativepronoun)))
tr_te$demonstrative_pronoun =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, demonstrative_pronoun)))
tr_te$indefinite_pronoun =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, indefinite_pronoun)))
tr_te$reflexive_pronoun =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, reflexive_pronoun)))
tr_te$interrogative_pronoun =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, interrogative_pronoun)))
tr_te$possessive_pronoun =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, possessive_pronoun)))
tr_te$subject_pronoun =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, subject_pronoun)))
tr_te$object_pronoun =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, object_pronoun)))
tr_te$do_proverb =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text,"do")))
tr_te$gerunds =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, gerunds)))
tr_te$by_passive <- laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, "by")))
tr_te$be_main_verb <- laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, "be")))
tr_te$existential_there <- laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, "there")))
tr_te$because <- laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, "because")))
tr_te$Concessive_adverbial_subordinators <- laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, c("although","though"))))
tr_te$Conditional_adverbial_subordinators <- laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, c("if","unless","until")))) 
tr_te$cojunct <- laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, cojunct)))
tr_te$hedges <-laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, hedges)))
tr_te$inflators <- laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, inflators)))
tr_te$discourse_marker <- laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, discourse_marker)))
tr_te$modal_possiblity =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, modal_possiblity)))
tr_te$modal_necessity =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, modal_necessity)))
tr_te$modal_predictive =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, modal_predictive)))
tr_te$verb_seem =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, verb_seem)))
tr_te$public_verb =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, public_verb)))
tr_te$private_verb =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, private_verb)))
tr_te$power_verbs =  laply(tr_te$question_text, function(question_text) sum(stri_detect_fixed(question_text, power_verbs)))
tr_te <- tr_te%>%  mutate(length = str_length(question_text),
                ncap = str_count(question_text, "[A-Z]"),
                nword = str_count(question_text, "\\w+"),
                totalredflagwords = Sexual_references+Age_dsicrim+Religion_racist+orientation_identity+mental_physicalhealth+Race_ethnicity+Google_banned_badwords+bad_words,
                modal_possiblity_ratio = modal_possiblity/nword,
                country_nationality_ratio = country_nationality/nword,
                bad_words_ratio = bad_words/nword,
                Google_banned_badwords_ratio = Google_banned_badwords/nword, 
                Sexual_references_ratio = Sexual_references/nword,
                Age_dsicrim_ratio = Age_dsicrim/nword,
                Race_ethnicity_ratio = Race_ethnicity/nword,
                mental_physicalhealth_ratio = mental_physicalhealth/nword,
                orientation_identity_ratio = orientation_identity/nword,
                Religion_racist_ratio = Religion_racist/nword,
                redflagtowordratio= totalredflagwords/nword,
                modal_necessity_ratio = modal_necessity/nword,
                modal_predictive_ratio = modal_predictive/nword,
                verb_seem_ratio = verb_seem/nword,
                public_verb_ratio = public_verb/nword,
                private_verb_ratio = private_verb/nword,
                power_verbs_ratio = power_verbs/nword,
                ncap_len = ncap / length,
                nnumers = str_count(question_text, fixed("!")),
                nexcl = str_count(question_text, fixed("!")),
                nquest = str_count(question_text, fixed("?")),
                npunct = str_count(question_text, "[[:punct:]]"),
                ncommas = str_count(question_text, fixed(",")),
                avgwordlen = length/nword,
                cojunct_ratio = cojunct/nword,
                hedges_ratio = hedges/nword,
                discourse_marker_ratio = discourse_marker/nword,
                inflatros_ratio = inflators/nword,
                do_proverb_ratio = do_proverb/nword,
                by_passive_ratio = by_passive/nword,
                be_main_verb_ratio = be_main_verb/nword,
                Derivational_suffix_ratio = Derivational_suffix/nword,
                Noun_of_action_suffix_ratio = Noun_of_action_suffix/nword,
                possessive_noun_ratio = possessive_noun/nword,
                past_simple_ratio = past_simple/nword,
                nominalization_ratio = nominalization/nword,
                present_tense_ratio = present_tense/nword,
                past_participate_ratio = past_participate/nword,
                frequency_adverb_ratio = frequency_adverb/nword,
                time_adverb_ratio = time_adverb/nword,
                place_adverb_ratio = place_adverb/nword,
                manner_adverb_ratio = manner_adverb/nword,
                degree_adverb_ratio = degree_adverb/nword,
                gerunds_ratio = gerunds/nword,
                object_pronoun_ratio = object_pronoun/nword,
                subject_pronoun_ratio = subject_pronoun/nword,
                possessive_pronoun_ratio = possessive_pronoun/nword,
                interrogative_pronoun_ratio = interrogative_pronoun/nword,
                reflexive_pronoun_ratio = reflexive_pronoun/nword,
                indefinite_pronoun_ratio = indefinite_pronoun/nword,
                demonstrative_pronoun_ratio = demonstrative_pronoun/nword,
                relativepronoun_ratio = relativepronoun/nword,
                thirdperson_pronoun_ratio = thirdperson_pronoun/nword,
                secondperson_pronoun_ratio = secondperson_pronoun/nword,
                firstperson_pronoun_ratio = firstperson_pronoun/nword,
                nsymb = str_count(question_text, "&|@|#|\\$|%|\\*|\\^"),
                nsmile = str_count(question_text, "((?::|;|=)(?:-)?(?:\\)|D|P))"),
                nsentence = str_count(question_text, fixed(".")))
library(keras)
library(kerasR)
library(text2vec)
library(purrrlyr)
library(reticulate)
library(textclean)
tr_te$question_text <- str_to_lower(tr_te$question_text) %>%
  str_replace_all("\\d", " number ") %>% replace_contraction() %>% replace_names(replacement= " Name ") %>% replace_non_ascii( replacement = "word") %>%  replace_white()

conv_fun <- function(x) iconv(x, "latin1", "ASCII", "")
library(purrrlyr)
tr_te <- tr_te %>% dmap_at(tr_te$question_text, conv_fun)
prep_fun <- tolower
tok_fun <- word_tokenizer
#tok_fun <- fit_text_tokenizer(tok_fun,tr_te$question_text)
it_complete <- text2vec::itoken(tr_te$question_text, 
                                preprocessor = prep_fun, 
                                tokenizer = tok_fun,ngram = c(ngram_min = 1L,
                                                              ngram_max = 3L),
                                ids = tr_te$qid,
                                progressbar = TRUE)
vocab <- text2vec::create_vocabulary(it_complete,ngram = c(ngram_min = 1L,ngram_max = 3L))
pruned_vocab = prune_vocabulary(vocab, term_count_min = 20,
                                doc_proportion_max = 0.7, doc_proportion_min = 0.01)
vectorizer <- vocab_vectorizer(pruned_vocab)
dtm_complete = create_dtm(it_complete, vectorizer,skip_grams_window = 2L)
tfidf = TfIdf$new()
# fit model to train data and transform train data with fitted model
dtm_complete = fit_transform(dtm_complete, tfidf)
rm(vocab,pruned_vocab); gc()
dtm1 <- data.frame(data.matrix(dtm_complete))
rm(dtm_complete); gc()
tr_te <- tr_te %>% cbind(dtm1)
rm(dtm1); gc()
cat("Prepare Data...\n")

library(xgboost)
tr <- tr_te[tri,]
tr_te <-tr_te[-tri,]
test <- subset(tr_te, qid %in% teid$qid)
val <- subset(tr_te, qid %in% valid$qid)
rm(tr_te); gc()
tr <- tr[,-c(1,2)]
cols <- colnames(tr)

test <- test[,-c(1,2)]
val <- val[,-c(1,2)]
val <- data.matrix(val)
test <- data.matrix(test)
dtest <- xgb.DMatrix(data = test)
rm(test); gc()
dval <- xgb.DMatrix(data = val, label = valy)
rm(val,valid); gc()
tr <- data.matrix(tr)
dtrain <- xgb.DMatrix(data = tr, label = y)
rm(tr); gc()
rm(trid,teid,fn); gc()


#---------------------------
cat("Training model...\n")
f1score_eval <- function(preds, dtrain) {
  labels <- getinfo(dtrain, "label")
  
  e_TP <- sum( (labels==1) & (preds >= 0.3) )
  e_FP <- sum( (labels==0) & (preds >= 0.3) )
  e_FN <- sum( (labels==1) & (preds < 0.3) )
  e_TN <- sum( (labels==0) & (preds < 0.3) )
  
  e_precision <- e_TP / (e_TP+e_FP)
  e_recall <- e_TP / (e_TP+e_FN)
  
  e_f1 <- 2*(e_precision*e_recall)/(e_precision+e_recall)
  
  return(list(metric = "f1-score", value = e_f1))
}

#---------------------------
cat("Training model...\n")
p <- list(objective = "binary:logistic",
          booster = "gbtree",
          eval_metric = f1score_eval,
          nthread = 4,
          eta = 0.05,
          #scale_pos_weight = 5,
          max_depth = 10,
          min_child_weight = 30,
          gamma = 1, max_delta_step = 7, reg_alpha = 0.5,
          subsample = 0.85,#scale_pos_weight= 30,
          colsample_bytree = 0.65,
          colsample_bylevel =0.65,
          alpha = 0, base_score=0.3, lambda = 0,
          #lambda = 0.735294,
          nrounds = 2000)
set.seed(0)
summary(y)
m_xgb <- xgb.train(p, dtrain, p$nrounds, list(val = dval), print_every_n = 10, early_stopping_rounds = 50,maximize=T)


xgb.importance(cols, model=m_xgb)%>% 
  xgb.plot.importance(top_n = 30)
round(m_xgb$best_score, 5)


pred <- data.frame(prediction=predict(m_xgb, dval),truth=valy)
sink(paste0("XGB",".txt"))
#Threshold search
best.f1 <- best.thresh <- 0
for(thresh in seq(0.1,0.5,0.01)){
  preds.thresh <- ifelse(pred$prediction >= thresh, 1, 0)
  y_val <- pred$truth
  Precision <- caret::precision(data = factor(preds.thresh), reference = factor(y_val), relevant = "1")
  Recall <- caret::recall(data = factor(preds.thresh), reference = factor(y_val), relevant = "1")
  f1.thresh <- 2 * (Precision * Recall)/(Precision + Recall)
  cat("Thresh = ", thresh, "     ","F1 = ", f1.thresh,"\n")
  best.thresh <- ifelse(f1.thresh >= best.f1, thresh,best.thresh)
  best.f1 <- ifelse(f1.thresh >= best.f1, f1.thresh,best.f1)
}
sink()

pred <- ifelse(predict(m_xgb, dtest)<best.thresh,0,1)
read_csv("../input/sample_submission.csv") %>%  
  mutate(qid = as.character(qid),
         prediction = pred) %>%
  write_csv("submission.csv")
