{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport random\nimport re\nfrom tqdm import tqdm\nimport string\nimport matplotlib.pyplot as plt\nfrom collections import defaultdict\nfrom nltk.corpus import stopwords\nimport tensorflow as tf\nfrom sklearn.feature_extraction.text import CountVectorizer\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-22T09:48:21.320690Z","iopub.execute_input":"2022-07-22T09:48:21.321458Z","iopub.status.idle":"2022-07-22T09:48:27.949039Z","shell.execute_reply.started":"2022-07-22T09:48:21.321330Z","shell.execute_reply":"2022-07-22T09:48:27.948079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As orientation we used: https://www.kaggle.com/code/shahules/basic-eda-cleaning-and-glove and https://www.kaggle.com/code/tuckerarrants/disaster-tweets-eda-glove-rnns-bert/notebook and https://www.kaggle.com/code/mitramir5/simple-bert-with-video","metadata":{}},{"cell_type":"code","source":"def seed_everything(seed):\n    os.environ['PYTHONHASHSEED']=str(seed)\n    tf.random.set_seed(seed)\n    np.random.seed(seed)\n    random.seed(seed)\n    \nseed_everything(42)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:27.953713Z","iopub.execute_input":"2022-07-22T09:48:27.956342Z","iopub.status.idle":"2022-07-22T09:48:27.964471Z","shell.execute_reply.started":"2022-07-22T09:48:27.956304Z","shell.execute_reply":"2022-07-22T09:48:27.963543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_path = '../input/nlp-getting-started'\ntrain_data = pd.read_csv(os.path.join(data_path,'train.csv'))\ntest_data = pd.read_csv(os.path.join(data_path,'test.csv'))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:27.970037Z","iopub.execute_input":"2022-07-22T09:48:27.972987Z","iopub.status.idle":"2022-07-22T09:48:28.054841Z","shell.execute_reply.started":"2022-07-22T09:48:27.972950Z","shell.execute_reply":"2022-07-22T09:48:28.053721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:28.061863Z","iopub.execute_input":"2022-07-22T09:48:28.064567Z","iopub.status.idle":"2022-07-22T09:48:28.093575Z","shell.execute_reply.started":"2022-07-22T09:48:28.064522Z","shell.execute_reply":"2022-07-22T09:48:28.092644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data[train_data['target'] == 0].head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:28.097915Z","iopub.execute_input":"2022-07-22T09:48:28.100341Z","iopub.status.idle":"2022-07-22T09:48:28.126045Z","shell.execute_reply.started":"2022-07-22T09:48:28.100300Z","shell.execute_reply":"2022-07-22T09:48:28.125083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:28.130490Z","iopub.execute_input":"2022-07-22T09:48:28.132882Z","iopub.status.idle":"2022-07-22T09:48:28.163484Z","shell.execute_reply.started":"2022-07-22T09:48:28.132841Z","shell.execute_reply":"2022-07-22T09:48:28.162459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:28.168485Z","iopub.execute_input":"2022-07-22T09:48:28.170918Z","iopub.status.idle":"2022-07-22T09:48:28.192230Z","shell.execute_reply.started":"2022-07-22T09:48:28.170878Z","shell.execute_reply":"2022-07-22T09:48:28.191249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#save ID\ntest_id = test_data['id']\n\n#drop from train and test\ntrain_data = train_data.drop(columns = 'location')\ntest_data = test_data.drop(columns = 'location')\n\n#fill missing with unknown\ntrain_data['keyword'] = train_data['keyword'].fillna('unknown')\ntest_data['keyword'] = test_data['keyword'].fillna('unknown')\n\n#add keyword to tweets\ntrain_data['text'] = train_data['text'] + ' ' + train_data['keyword']\ntest_data['text'] = test_data['text'] + ' ' + test_data['keyword']\n\n#drop keyword from train and test\ncolumns = {'keyword'}\ntrain_data = train_data.drop(columns = columns)\ntest_data = test_data.drop(columns = columns)\n\n#combine so we work smarter, not harder\ntotal_data = train_data.append(test_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:28.196865Z","iopub.execute_input":"2022-07-22T09:48:28.199228Z","iopub.status.idle":"2022-07-22T09:48:28.231713Z","shell.execute_reply.started":"2022-07-22T09:48:28.199172Z","shell.execute_reply":"2022-07-22T09:48:28.230725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counts = train_data.target.value_counts()\nsns.barplot(x=counts.index,y=counts)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:28.236637Z","iopub.execute_input":"2022-07-22T09:48:28.239016Z","iopub.status.idle":"2022-07-22T09:48:28.445646Z","shell.execute_reply.started":"2022-07-22T09:48:28.238976Z","shell.execute_reply":"2022-07-22T09:48:28.444572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"disaster_tweet_characters = train_data[train_data[\"target\"] == 1]['text'].str.len()\nnon_disaster_tweet_characters = train_data[train_data[\"target\"] == 0]['text'].str.len()\nfig,(ax1,ax2)=plt.subplots(1,2,figsize=(10,5))\nax1.hist(disaster_tweet_characters)\nax1.set_title('Disaster tweets')\nax2.hist(non_disaster_tweet_characters,color='red')\nax2.set_title('Non-Disaster tweets')\nfig.suptitle('Characters in tweets')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:28.450696Z","iopub.execute_input":"2022-07-22T09:48:28.451656Z","iopub.status.idle":"2022-07-22T09:48:28.821859Z","shell.execute_reply.started":"2022-07-22T09:48:28.451617Z","shell.execute_reply":"2022-07-22T09:48:28.820852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"disaster_tweet_word_count = train_data[train_data[\"target\"] == 1]['text'].str.split().map(lambda x: len(x))\nnon_disaster_tweet_word_count = train_data[train_data[\"target\"] == 0]['text'].str.split().map(lambda x: len(x))\nfig,(ax1,ax2)=plt.subplots(1,2,figsize=(10,5))\nax1.hist(disaster_tweet_word_count)\nax1.set_title('Disaster tweets')\nax2.hist(non_disaster_tweet_word_count,color='red')\nax2.set_title('Non-Disaster tweets')\nfig.suptitle('Wordcount in tweets')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:28.826419Z","iopub.execute_input":"2022-07-22T09:48:28.826818Z","iopub.status.idle":"2022-07-22T09:48:29.233420Z","shell.execute_reply.started":"2022-07-22T09:48:28.826791Z","shell.execute_reply":"2022-07-22T09:48:29.232249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_corpus(target):\n    corpus = []\n    for tweet in train_data[train_data[\"target\"] == target]['text'].str.split():\n        for word in tweet:\n            corpus.append(word.lower()) #Put words to lowercase as well\n    return corpus","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:29.234639Z","iopub.execute_input":"2022-07-22T09:48:29.235245Z","iopub.status.idle":"2022-07-22T09:48:29.242289Z","shell.execute_reply.started":"2022-07-22T09:48:29.235208Z","shell.execute_reply":"2022-07-22T09:48:29.241243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stopws = list(stopwords.words('english')) #list of words that are not important for the true meaning\ndef get_non_stop_word_count(target): #Counts appearances of non stopwords\n    corpus = get_corpus(target)\n    dic=defaultdict(int) #dict that returns 0 if key does not exists\n    for word in corpus:\n        if word not in stopws:\n            dic[word] += 1\n    return dic","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:29.243928Z","iopub.execute_input":"2022-07-22T09:48:29.244350Z","iopub.status.idle":"2022-07-22T09:48:29.256989Z","shell.execute_reply.started":"2022-07-22T09:48:29.244294Z","shell.execute_reply":"2022-07-22T09:48:29.255940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dic = get_non_stop_word_count(0)\ntop=sorted(dic.items(), key=lambda x:x[1],reverse=True)[:10]\nx,y=zip(*top)\nplt.bar(x,y)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:29.258581Z","iopub.execute_input":"2022-07-22T09:48:29.260402Z","iopub.status.idle":"2022-07-22T09:48:29.594855Z","shell.execute_reply.started":"2022-07-22T09:48:29.260374Z","shell.execute_reply":"2022-07-22T09:48:29.593900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dic = get_non_stop_word_count(1)\ntop=sorted(dic.items(), key=lambda x:x[1],reverse=True)[:10]\nx,y=zip(*top)\nplt.bar(x,y)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:29.596354Z","iopub.execute_input":"2022-07-22T09:48:29.596934Z","iopub.status.idle":"2022-07-22T09:48:29.905385Z","shell.execute_reply.started":"2022-07-22T09:48:29.596895Z","shell.execute_reply":"2022-07-22T09:48:29.904378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Cleaning","metadata":{}},{"cell_type":"code","source":"total_data['text'] = total_data['text'].map(lambda x: x.lower())","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:29.906924Z","iopub.execute_input":"2022-07-22T09:48:29.907541Z","iopub.status.idle":"2022-07-22T09:48:29.919486Z","shell.execute_reply.started":"2022-07-22T09:48:29.907500Z","shell.execute_reply":"2022-07-22T09:48:29.918520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_URL(text):\n    url = re.compile(r'https?://\\S+|www\\.\\S+')\n    return url.sub(r'',text)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:29.922079Z","iopub.execute_input":"2022-07-22T09:48:29.923436Z","iopub.status.idle":"2022-07-22T09:48:29.928459Z","shell.execute_reply.started":"2022-07-22T09:48:29.923397Z","shell.execute_reply":"2022-07-22T09:48:29.926891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_data['text'] = total_data['text'].map(lambda x: remove_URL(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:29.930167Z","iopub.execute_input":"2022-07-22T09:48:29.930560Z","iopub.status.idle":"2022-07-22T09:48:29.985406Z","shell.execute_reply.started":"2022-07-22T09:48:29.930523Z","shell.execute_reply":"2022-07-22T09:48:29.984467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_html(text):\n    html=re.compile(r'<.*?>')\n    return html.sub(r'',text)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:29.986852Z","iopub.execute_input":"2022-07-22T09:48:29.987210Z","iopub.status.idle":"2022-07-22T09:48:29.992068Z","shell.execute_reply.started":"2022-07-22T09:48:29.987149Z","shell.execute_reply":"2022-07-22T09:48:29.991043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_data['text'] = total_data['text'].map(lambda x: remove_html(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:29.993623Z","iopub.execute_input":"2022-07-22T09:48:29.994346Z","iopub.status.idle":"2022-07-22T09:48:30.017947Z","shell.execute_reply.started":"2022-07-22T09:48:29.994260Z","shell.execute_reply":"2022-07-22T09:48:30.016799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reference : https://gist.github.com/slowkow/7a7f61f495e3dbb7e3d767f97bd7304b\ndef remove_emoji(text):\n    emoji_pattern = re.compile(\"[\"\n                           u\"\\U0001F600-\\U0001F64F\"  # emoticons\n                           u\"\\U0001F300-\\U0001F5FF\"  # symbols & pictographs\n                           u\"\\U0001F680-\\U0001F6FF\"  # transport & map symbols\n                           u\"\\U0001F1E0-\\U0001F1FF\"  # flags (iOS)\n                           u\"\\U00002702-\\U000027B0\"\n                           u\"\\U000024C2-\\U0001F251\"\n                           \"]+\", flags=re.UNICODE)\n    return emoji_pattern.sub(r'', text)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:30.019623Z","iopub.execute_input":"2022-07-22T09:48:30.019984Z","iopub.status.idle":"2022-07-22T09:48:30.029212Z","shell.execute_reply.started":"2022-07-22T09:48:30.019950Z","shell.execute_reply":"2022-07-22T09:48:30.027994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_data['text'] = total_data['text'].map(lambda x: remove_emoji(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:30.030895Z","iopub.execute_input":"2022-07-22T09:48:30.031370Z","iopub.status.idle":"2022-07-22T09:48:30.121863Z","shell.execute_reply.started":"2022-07-22T09:48:30.031333Z","shell.execute_reply":"2022-07-22T09:48:30.120867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean(tweet):\n\n    #correct some acronyms while we are at it\n    tweet = re.sub(r\"tnwx\", \"Tennessee Weather\", tweet)\n    tweet = re.sub(r\"azwx\", \"Arizona Weather\", tweet)  \n    tweet = re.sub(r\"alwx\", \"Alabama Weather\", tweet)\n    tweet = re.sub(r\"wordpressdotcom\", \"wordpress\", tweet)      \n    tweet = re.sub(r\"gawx\", \"Georgia Weather\", tweet)  \n    tweet = re.sub(r\"scwx\", \"South Carolina Weather\", tweet)  \n    tweet = re.sub(r\"cawx\", \"California Weather\", tweet)\n    tweet = re.sub(r\"usNWSgov\", \"United States National Weather Service\", tweet) \n    tweet = re.sub(r\"MH370\", \"Malaysia Airlines Flight 370\", tweet)\n    tweet = re.sub(r\"okwx\", \"Oklahoma City Weather\", tweet)\n    tweet = re.sub(r\"arwx\", \"Arkansas Weather\", tweet)  \n    tweet = re.sub(r\"lmao\", \"laughing my ass off\", tweet)  \n    tweet = re.sub(r\"amirite\", \"am I right\", tweet)\n    \n    #and some typos/abbreviations\n    tweet = re.sub(r\"w/e\", \"whatever\", tweet)\n    tweet = re.sub(r\"w/\", \"with\", tweet)\n    tweet = re.sub(r\"USAgov\", \"USA government\", tweet)\n    tweet = re.sub(r\"recentlu\", \"recently\", tweet)\n    tweet = re.sub(r\"Ph0tos\", \"Photos\", tweet)\n    tweet = re.sub(r\"exp0sed\", \"exposed\", tweet)\n    tweet = re.sub(r\"<3\", \"love\", tweet)\n    tweet = re.sub(r\"amageddon\", \"armageddon\", tweet)\n    tweet = re.sub(r\"Trfc\", \"Traffic\", tweet)\n    tweet = re.sub(r\"WindStorm\", \"Wind Storm\", tweet)\n    tweet = re.sub(r\"16yr\", \"16 year\", tweet)\n    tweet = re.sub(r\"TRAUMATISED\", \"traumatized\", tweet)\n    \n    #hashtags and usernames\n    tweet = re.sub(r\"IranDeal\", \"Iran Deal\", tweet)\n    tweet = re.sub(r\"ArianaGrande\", \"Ariana Grande\", tweet)\n    tweet = re.sub(r\"camilacabello97\", \"camila cabello\", tweet) \n    tweet = re.sub(r\"RondaRousey\", \"Ronda Rousey\", tweet)     \n    tweet = re.sub(r\"MTVHottest\", \"MTV Hottest\", tweet)\n    tweet = re.sub(r\"TrapMusic\", \"Trap Music\", tweet)\n    tweet = re.sub(r\"ProphetMuhammad\", \"Prophet Muhammad\", tweet)\n    tweet = re.sub(r\"PantherAttack\", \"Panther Attack\", tweet)\n    tweet = re.sub(r\"StrategicPatience\", \"Strategic Patience\", tweet)\n    tweet = re.sub(r\"socialnews\", \"social news\", tweet)\n    tweet = re.sub(r\"IDPs:\", \"Internally Displaced People :\", tweet)\n    tweet = re.sub(r\"ArtistsUnited\", \"Artists United\", tweet)\n    tweet = re.sub(r\"ClaytonBryant\", \"Clayton Bryant\", tweet)\n    tweet = re.sub(r\"jimmyfallon\", \"jimmy fallon\", tweet)\n    tweet = re.sub(r\"justinbieber\", \"justin bieber\", tweet)  \n    tweet = re.sub(r\"Time2015\", \"Time 2015\", tweet)\n    tweet = re.sub(r\"djicemoon\", \"dj icemoon\", tweet)\n    tweet = re.sub(r\"LivingSafely\", \"Living Safely\", tweet)\n    tweet = re.sub(r\"FIFA16\", \"Fifa 2016\", tweet)\n    tweet = re.sub(r\"thisiswhywecanthavenicethings\", \"this is why we cannot have nice things\", tweet)\n    tweet = re.sub(r\"bbcnews\", \"bbc news\", tweet)\n    tweet = re.sub(r\"UndergroundRailraod\", \"Underground Railraod\", tweet)\n    tweet = re.sub(r\"c4news\", \"c4 news\", tweet)\n    tweet = re.sub(r\"MUDSLIDE\", \"mudslide\", tweet)\n    tweet = re.sub(r\"NoSurrender\", \"No Surrender\", tweet)\n    tweet = re.sub(r\"NotExplained\", \"Not Explained\", tweet)\n    tweet = re.sub(r\"greatbritishbakeoff\", \"great british bake off\", tweet)\n    tweet = re.sub(r\"LondonFire\", \"London Fire\", tweet)\n    tweet = re.sub(r\"KOTAWeather\", \"KOTA Weather\", tweet)\n    tweet = re.sub(r\"LuchaUnderground\", \"Lucha Underground\", tweet)\n    tweet = re.sub(r\"KOIN6News\", \"KOIN 6 News\", tweet)\n    tweet = re.sub(r\"LiveOnK2\", \"Live On K2\", tweet)\n    tweet = re.sub(r\"9NewsGoldCoast\", \"9 News Gold Coast\", tweet)\n    tweet = re.sub(r\"nikeplus\", \"nike plus\", tweet)\n    tweet = re.sub(r\"david_cameron\", \"David Cameron\", tweet)\n    tweet = re.sub(r\"peterjukes\", \"Peter Jukes\", tweet)\n    tweet = re.sub(r\"MikeParrActor\", \"Michael Parr\", tweet)\n    tweet = re.sub(r\"4PlayThursdays\", \"Foreplay Thursdays\", tweet)\n    tweet = re.sub(r\"TGF2015\", \"Tontitown Grape Festival\", tweet)\n    tweet = re.sub(r\"realmandyrain\", \"Mandy Rain\", tweet)\n    tweet = re.sub(r\"GraysonDolan\", \"Grayson Dolan\", tweet)\n    tweet = re.sub(r\"ApolloBrown\", \"Apollo Brown\", tweet)\n    tweet = re.sub(r\"saddlebrooke\", \"Saddlebrooke\", tweet)\n    tweet = re.sub(r\"TontitownGrape\", \"Tontitown Grape\", tweet)\n    tweet = re.sub(r\"AbbsWinston\", \"Abbs Winston\", tweet)\n    tweet = re.sub(r\"ShaunKing\", \"Shaun King\", tweet)\n    tweet = re.sub(r\"MeekMill\", \"Meek Mill\", tweet)\n    tweet = re.sub(r\"TornadoGiveaway\", \"Tornado Giveaway\", tweet)\n    tweet = re.sub(r\"GRupdates\", \"GR updates\", tweet)\n    tweet = re.sub(r\"SouthDowns\", \"South Downs\", tweet)\n    tweet = re.sub(r\"braininjury\", \"brain injury\", tweet)\n    tweet = re.sub(r\"auspol\", \"Australian politics\", tweet)\n    tweet = re.sub(r\"PlannedParenthood\", \"Planned Parenthood\", tweet)\n    tweet = re.sub(r\"calgaryweather\", \"Calgary Weather\", tweet)\n    tweet = re.sub(r\"weallheartonedirection\", \"we all heart one direction\", tweet)\n    tweet = re.sub(r\"edsheeran\", \"Ed Sheeran\", tweet)\n    tweet = re.sub(r\"TrueHeroes\", \"True Heroes\", tweet)\n    tweet = re.sub(r\"ComplexMag\", \"Complex Magazine\", tweet)\n    tweet = re.sub(r\"TheAdvocateMag\", \"The Advocate Magazine\", tweet)\n    tweet = re.sub(r\"CityofCalgary\", \"City of Calgary\", tweet)\n    tweet = re.sub(r\"EbolaOutbreak\", \"Ebola Outbreak\", tweet)\n    tweet = re.sub(r\"SummerFate\", \"Summer Fate\", tweet)\n    tweet = re.sub(r\"RAmag\", \"Royal Academy Magazine\", tweet)\n    tweet = re.sub(r\"offers2go\", \"offers to go\", tweet)\n    tweet = re.sub(r\"ModiMinistry\", \"Modi Ministry\", tweet)\n    tweet = re.sub(r\"TAXIWAYS\", \"taxi ways\", tweet)\n    tweet = re.sub(r\"Calum5SOS\", \"Calum Hood\", tweet)\n    tweet = re.sub(r\"JamesMelville\", \"James Melville\", tweet)\n    tweet = re.sub(r\"JamaicaObserver\", \"Jamaica Observer\", tweet)\n    tweet = re.sub(r\"TweetLikeItsSeptember11th2001\", \"Tweet like it is september 11th 2001\", tweet)\n    tweet = re.sub(r\"cbplawyers\", \"cbp lawyers\", tweet)\n    tweet = re.sub(r\"fewmoretweets\", \"few more tweets\", tweet)\n    tweet = re.sub(r\"BlackLivesMatter\", \"Black Lives Matter\", tweet)\n    tweet = re.sub(r\"NASAHurricane\", \"NASA Hurricane\", tweet)\n    tweet = re.sub(r\"onlinecommunities\", \"online communities\", tweet)\n    tweet = re.sub(r\"humanconsumption\", \"human consumption\", tweet)\n    tweet = re.sub(r\"Typhoon-Devastated\", \"Typhoon Devastated\", tweet)\n    tweet = re.sub(r\"Meat-Loving\", \"Meat Loving\", tweet)\n    tweet = re.sub(r\"facialabuse\", \"facial abuse\", tweet)\n    tweet = re.sub(r\"LakeCounty\", \"Lake County\", tweet)\n    tweet = re.sub(r\"BeingAuthor\", \"Being Author\", tweet)\n    tweet = re.sub(r\"withheavenly\", \"with heavenly\", tweet)\n    tweet = re.sub(r\"thankU\", \"thank you\", tweet)\n    tweet = re.sub(r\"iTunesMusic\", \"iTunes Music\", tweet)\n    tweet = re.sub(r\"OffensiveContent\", \"Offensive Content\", tweet)\n    tweet = re.sub(r\"WorstSummerJob\", \"Worst Summer Job\", tweet)\n    tweet = re.sub(r\"HarryBeCareful\", \"Harry Be Careful\", tweet)\n    tweet = re.sub(r\"NASASolarSystem\", \"NASA Solar System\", tweet)\n    tweet = re.sub(r\"animalrescue\", \"animal rescue\", tweet)\n    tweet = re.sub(r\"KurtSchlichter\", \"Kurt Schlichter\", tweet)\n    tweet = re.sub(r\"Throwingknifes\", \"Throwing knives\", tweet)\n    tweet = re.sub(r\"GodsLove\", \"God's Love\", tweet)\n    tweet = re.sub(r\"bookboost\", \"book boost\", tweet)\n    tweet = re.sub(r\"ibooklove\", \"I book love\", tweet)\n    tweet = re.sub(r\"NestleIndia\", \"Nestle India\", tweet)\n    tweet = re.sub(r\"realDonaldTrump\", \"Donald Trump\", tweet)\n    tweet = re.sub(r\"DavidVonderhaar\", \"David Vonderhaar\", tweet)\n    tweet = re.sub(r\"CecilTheLion\", \"Cecil The Lion\", tweet)\n    tweet = re.sub(r\"weathernetwork\", \"weather network\", tweet)\n    tweet = re.sub(r\"GOPDebate\", \"GOP Debate\", tweet)\n    tweet = re.sub(r\"RickPerry\", \"Rick Perry\", tweet)\n    tweet = re.sub(r\"frontpage\", \"front page\", tweet)\n    tweet = re.sub(r\"NewsInTweets\", \"News In Tweets\", tweet)\n    tweet = re.sub(r\"ViralSpell\", \"Viral Spell\", tweet)\n    tweet = re.sub(r\"til_now\", \"until now\", tweet)\n    tweet = re.sub(r\"volcanoinRussia\", \"volcano in Russia\", tweet)\n    tweet = re.sub(r\"ZippedNews\", \"Zipped News\", tweet)\n    tweet = re.sub(r\"MicheleBachman\", \"Michele Bachman\", tweet)\n    tweet = re.sub(r\"53inch\", \"53 inch\", tweet)\n    tweet = re.sub(r\"KerrickTrial\", \"Kerrick Trial\", tweet)\n    tweet = re.sub(r\"abstorm\", \"Alberta Storm\", tweet)\n    tweet = re.sub(r\"Beyhive\", \"Beyonce hive\", tweet)\n    tweet = re.sub(r\"RockyFire\", \"Rocky Fire\", tweet)\n    tweet = re.sub(r\"Listen/Buy\", \"Listen / Buy\", tweet)\n    tweet = re.sub(r\"ArtistsUnited\", \"Artists United\", tweet)\n    tweet = re.sub(r\"ENGvAUS\", \"England vs Australia\", tweet)\n    tweet = re.sub(r\"ScottWalker\", \"Scott Walker\", tweet)\n\n    return tweet","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:30.123786Z","iopub.execute_input":"2022-07-22T09:48:30.124322Z","iopub.status.idle":"2022-07-22T09:48:30.169356Z","shell.execute_reply.started":"2022-07-22T09:48:30.124284Z","shell.execute_reply":"2022-07-22T09:48:30.168370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_data['text'] = total_data['text'].map(lambda x: clean(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:30.170775Z","iopub.execute_input":"2022-07-22T09:48:30.171143Z","iopub.status.idle":"2022-07-22T09:48:31.950615Z","shell.execute_reply.started":"2022-07-22T09:48:30.171106Z","shell.execute_reply":"2022-07-22T09:48:31.949493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Getting Glove Embeddings","metadata":{}},{"cell_type":"markdown","source":"Actually we don't use this for BERT","metadata":{}},{"cell_type":"code","source":"# glove_embeddings = {}\n# with open('../input/glove6b/glove.6B.300d.txt','r') as f:\n#     for line in tqdm(f):\n#         values = line.split()\n#         word = values[0]\n#         embedding = np.asarray(values[1:],dtype='float32')\n#         glove_embeddings[word] = embedding\n# GLOVE_DIM = 300","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:31.952018Z","iopub.execute_input":"2022-07-22T09:48:31.952382Z","iopub.status.idle":"2022-07-22T09:48:31.957868Z","shell.execute_reply.started":"2022-07-22T09:48:31.952347Z","shell.execute_reply":"2022-07-22T09:48:31.956477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from keras.preprocessing.text import Tokenizer\n\n# #Give each unique word a token\n# tokenizer = Tokenizer()\n# tokenizer.fit_on_texts(total_data['text'])\n# word_indices = tokenizer.word_index","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:31.959705Z","iopub.execute_input":"2022-07-22T09:48:31.959961Z","iopub.status.idle":"2022-07-22T09:48:31.968670Z","shell.execute_reply.started":"2022-07-22T09:48:31.959938Z","shell.execute_reply":"2022-07-22T09:48:31.966579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# embedded_tokens = np.zeros((len(word_indices),GLOVE_DIM),dtype='float32')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:31.969679Z","iopub.execute_input":"2022-07-22T09:48:31.970401Z","iopub.status.idle":"2022-07-22T09:48:31.976153Z","shell.execute_reply.started":"2022-07-22T09:48:31.970364Z","shell.execute_reply":"2022-07-22T09:48:31.975132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for (word,index) in tqdm(word_indices.items()):\n#     embedding = glove_embeddings.get(word)\n#     if embedding is not None: #Otherwise we just keep the embedding to be zero\n#         embedded_tokens[index] = embedding","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:31.982255Z","iopub.execute_input":"2022-07-22T09:48:31.983122Z","iopub.status.idle":"2022-07-22T09:48:31.987513Z","shell.execute_reply.started":"2022-07-22T09:48:31.983089Z","shell.execute_reply":"2022-07-22T09:48:31.986376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(embedded_tokens)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:31.988832Z","iopub.execute_input":"2022-07-22T09:48:31.989835Z","iopub.status.idle":"2022-07-22T09:48:31.996918Z","shell.execute_reply.started":"2022-07-22T09:48:31.989791Z","shell.execute_reply":"2022-07-22T09:48:31.996035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# BERT","metadata":{}},{"cell_type":"code","source":"!pip install transformers --quiet","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:32.000214Z","iopub.execute_input":"2022-07-22T09:48:32.001081Z","iopub.status.idle":"2022-07-22T09:48:42.811288Z","shell.execute_reply.started":"2022-07-22T09:48:32.001042Z","shell.execute_reply":"2022-07-22T09:48:42.810103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import BertTokenizer\n\ntokenizer = BertTokenizer.from_pretrained(\"bert-base-cased\")","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:42.813136Z","iopub.execute_input":"2022-07-22T09:48:42.813855Z","iopub.status.idle":"2022-07-22T09:48:47.874834Z","shell.execute_reply.started":"2022-07-22T09:48:42.813813Z","shell.execute_reply":"2022-07-22T09:48:47.873761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import TFBertModel\n\nbert_layer = TFBertModel.from_pretrained(\"bert-base-cased\")","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:48:47.876585Z","iopub.execute_input":"2022-07-22T09:48:47.876968Z","iopub.status.idle":"2022-07-22T09:49:19.493735Z","shell.execute_reply.started":"2022-07-22T09:48:47.876928Z","shell.execute_reply":"2022-07-22T09:49:19.492731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TweetClassifier:\n    \n    def __init__(self, tokenizer, bert_layer, max_len, lr = 0.0001,\n                 epochs = 15, batch_size = 32,\n                 activation = 'sigmoid', optimizer = 'SGD',\n                 beta_1=0.9, beta_2=0.999, epsilon=1e-07,\n                 metrics = 'accuracy', loss = 'binary_crossentropy'):\n        \n        self.lr = lr\n        self.epochs = epochs\n        self.max_len = max_len\n        self.batch_size = batch_size\n        self.tokenizer = tokenizer\n        self.bert_layer = bert_layer\n        \n\n        self.activation = activation\n        self.optimizer = optimizer\n        \n        self.beta_1 = beta_1\n        self.beta_2 = beta_2\n        self.epsilon = epsilon\n        \n        self.metrics = metrics\n        self.loss = loss\n\n    def encode(self, texts):\n        \"\"\"\n        Encode texts using BERTs tokenization and return input_ids as well as attention masks for all samples.\n        \"\"\"\n        input_ids = []\n        attention_masks = []\n        \n        for text in texts:\n            encoded = self.tokenizer.encode_plus(text,\n                                        add_special_tokens=True,\n                                        max_length=self.max_len,\n                                        pad_to_max_length=True,\n                                        return_attention_mask=True)\n            input_ids.append(encoded['input_ids'])\n            attention_masks.append(encoded['attention_mask'])\n            \n        return np.array(input_ids), np.array(attention_masks)\n    \n    def build_model(self):\n        input_ids = tf.keras.Input(shape=(self.max_len,),dtype='int32')\n        attention_masks = tf.keras.Input(shape=(self.max_len,),dtype='int32')\n        \n        output = self.bert_layer([input_ids, attention_masks])\n        output = output[1] #choose only last hidden-state\n        \n        #add final node for binary classification\n        output = tf.keras.layers.Dense(1,activation=self.activation)(output)\n        \n        model = tf.keras.models.Model(inputs=[input_ids,attention_masks],outputs=output)\n        \n        model.compile(loss = self.loss, optimizer = self.optimizer, metrics = [self.metrics])\n        \n        return model\n    \n    def train(self, x):    \n        checkpoint = tf.keras.callbacks.ModelCheckpoint('model.h5',monitor='val_loss',save_best_only=True,save_weights_only = True)\n        \n        model = self.build_model()\n        \n        X = self.encode(x['text'])\n        Y = x['target']\n        \n        model.fit(X,Y,shuffle=True,validation_split=0.2,batch_size=self.batch_size,epochs=self.epochs,callbacks=[checkpoint])\n        \n        self.model = model\n                \n    def predict(self, x):\n        X_test_encoded = self.encode(x['text'])\n        self.model.load_weights('model.h5') \n        y_pred = self.model.predict(X_test_encoded)\n        \n        return y_pred","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:54:45.723871Z","iopub.execute_input":"2022-07-22T09:54:45.724287Z","iopub.status.idle":"2022-07-22T09:54:45.742367Z","shell.execute_reply.started":"2022-07-22T09:54:45.724250Z","shell.execute_reply":"2022-07-22T09:54:45.741332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classifier = TweetClassifier(tokenizer = tokenizer, bert_layer = bert_layer,\n                              max_len = 60, lr = 0.0001,\n                              epochs = 3,  activation = 'sigmoid',\n                              batch_size = 32,optimizer = 'SGD',\n                              beta_1=0.9, beta_2=0.999, epsilon=1e-07)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:54:47.740845Z","iopub.execute_input":"2022-07-22T09:54:47.741250Z","iopub.status.idle":"2022-07-22T09:54:47.747269Z","shell.execute_reply.started":"2022-07-22T09:54:47.741210Z","shell.execute_reply":"2022-07-22T09:54:47.745981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classifier.train(total_data[:len(train_data)])","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:54:50.602367Z","iopub.execute_input":"2022-07-22T09:54:50.602744Z","iopub.status.idle":"2022-07-22T09:57:39.309505Z","shell.execute_reply.started":"2022-07-22T09:54:50.602713Z","shell.execute_reply":"2022-07-22T09:57:39.308507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = np.round(classifier.predict(total_data[len(train_data):]))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:57:39.312891Z","iopub.execute_input":"2022-07-22T09:57:39.313811Z","iopub.status.idle":"2022-07-22T09:57:53.038157Z","shell.execute_reply.started":"2022-07-22T09:57:39.313767Z","shell.execute_reply":"2022-07-22T09:57:53.037114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Submission\nsample_sub = pd.read_csv('../input/nlp-getting-started/sample_submission.csv')\nids = sample_sub.id\nfinal_submission = pd.DataFrame(np.c_[ids, y_pred.astype('int')], columns = ['id', 'target'])\nfinal_submission.to_csv('final_submission.csv', index = False)\nfinal_submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T10:15:19.966413Z","iopub.execute_input":"2022-07-22T10:15:19.967461Z","iopub.status.idle":"2022-07-22T10:15:19.987990Z","shell.execute_reply.started":"2022-07-22T10:15:19.967412Z","shell.execute_reply":"2022-07-22T10:15:19.986904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}