{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h1 style='background:#FF008C; font-size:36px; color:white; padding-top:25px;'><center> <b> Tweets Classification with BERT </b> </center> </h1>","metadata":{}},{"cell_type":"markdown","source":"![](http://miro.medium.com/max/688/0*B8VDCnh8qBnwUuM4)","metadata":{}},{"cell_type":"markdown","source":"<h2 style='background:#FF6F00; color:white; padding:15px 0;'> Importing Required Libraries && tensorflow_text is must for Loading Preprocessor from Tensorflow hub. </h2>","metadata":{}},{"cell_type":"code","source":"!pip install tensorflow_text\nimport tensorflow_text as text","metadata":{"execution":{"iopub.status.busy":"2022-07-07T15:37:52.664755Z","iopub.execute_input":"2022-07-07T15:37:52.665190Z","iopub.status.idle":"2022-07-07T15:38:02.218396Z","shell.execute_reply.started":"2022-07-07T15:37:52.665154Z","shell.execute_reply":"2022-07-07T15:38:02.217410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\nimport string\nimport numpy as np\nimport pandas as pd\n\nfrom bs4 import BeautifulSoup\n\nimport plotly.express as px\n\nimport tensorflow as tf\nimport tensorflow_hub as hub","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-07T15:39:43.195409Z","iopub.execute_input":"2022-07-07T15:39:43.196290Z","iopub.status.idle":"2022-07-07T15:39:43.201768Z","shell.execute_reply.started":"2022-07-07T15:39:43.196251Z","shell.execute_reply":"2022-07-07T15:39:43.200599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"![](http://miro.medium.com/max/1200/1*fnfos3Z_xWdK9GdxX_pGxQ.jpeg)","metadata":{}},{"cell_type":"markdown","source":" <h1 style='color:#ff8300; font-size:40px;'> <center> Bidirectional Encoder Representations from Transformers </center> </h1>","metadata":{}},{"cell_type":"markdown","source":"<div style='font-size:24px;'>\n    \n> BERT provides dense vector representations for natural language by using a deep, pre-trained neural network with the Transformer architecture. \n\n> Along with the Encoder, There is need to load the Bert Preprocessor which take cares of all of the text processing automatically that is being directly feed to the network.    \n    \n> ***BERT is the Model which is being used by Google in the Google Search Engine.***\n    \n> It is being trained on millions of wikipedia and other popular websites texts.\n    \n> Tensorflow provide API to load these Pretrained Models and Layers.\n    \n</div>","metadata":{}},{"cell_type":"code","source":"bert_preprocessor = hub.KerasLayer('https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3')\nbert_encoder = hub.KerasLayer('https://tfhub.dev/tensorflow/bert_en_uncased_L-12_H-768_A-12/4')","metadata":{"execution":{"iopub.status.busy":"2022-07-07T15:39:46.296858Z","iopub.execute_input":"2022-07-07T15:39:46.297367Z","iopub.status.idle":"2022-07-07T15:40:00.882302Z","shell.execute_reply.started":"2022-07-07T15:39:46.297334Z","shell.execute_reply":"2022-07-07T15:40:00.881446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2 style='background:#FF6F00; color:white; padding:15px 0;'> Configuring TPU or GPU</h2>","metadata":{}},{"cell_type":"markdown","source":"![](https://miro.medium.com/max/640/1*Cf7HHO-ozFddK_qWT1U_vg.png)","metadata":{}},{"cell_type":"code","source":"try:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver() \n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    strategy = tf.distribute.get_strategy() \n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T15:40:00.884088Z","iopub.execute_input":"2022-07-07T15:40:00.884750Z","iopub.status.idle":"2022-07-07T15:40:00.893095Z","shell.execute_reply.started":"2022-07-07T15:40:00.884711Z","shell.execute_reply":"2022-07-07T15:40:00.892290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2 style='background:#FF6F00; color:white; padding:15px 0;'> Setting up Dataset using Cloud Bucket (for TPU) ☁️ </h2>","metadata":{}},{"cell_type":"code","source":"from kaggle_datasets import KaggleDatasets\nGCS_DS_PATH = KaggleDatasets().get_gcs_path('nlp-getting-started')\nprint(GCS_DS_PATH) ","metadata":{"execution":{"iopub.status.busy":"2022-07-07T15:40:00.894365Z","iopub.execute_input":"2022-07-07T15:40:00.895204Z","iopub.status.idle":"2022-07-07T15:40:01.666110Z","shell.execute_reply.started":"2022-07-07T15:40:00.895169Z","shell.execute_reply":"2022-07-07T15:40:01.665258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nTRAINING_FILENAMES = tf.io.gfile.glob(GCS_DS_PATH + '/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-07T15:40:01.669853Z","iopub.execute_input":"2022-07-07T15:40:01.672955Z","iopub.status.idle":"2022-07-07T15:40:02.065467Z","shell.execute_reply.started":"2022-07-07T15:40:01.672906Z","shell.execute_reply":"2022-07-07T15:40:02.064610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_rows', None)\npd.set_option('display.max_columns', None)\nraw_data = pd.read_csv(TRAINING_FILENAMES[0])","metadata":{"execution":{"iopub.status.busy":"2022-07-07T15:40:02.067138Z","iopub.execute_input":"2022-07-07T15:40:02.067790Z","iopub.status.idle":"2022-07-07T15:40:03.579176Z","shell.execute_reply.started":"2022-07-07T15:40:02.067752Z","shell.execute_reply":"2022-07-07T15:40:03.578380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_data.head().style.set_table_styles(\n    [{'selector': 'tr:hover',\n      'props': [('background-color', 'green')]}]\n).set_properties(**{\n    'font-size': '15pt',})","metadata":{"execution":{"iopub.status.busy":"2022-07-07T15:40:03.580554Z","iopub.execute_input":"2022-07-07T15:40:03.580894Z","iopub.status.idle":"2022-07-07T15:40:03.596567Z","shell.execute_reply.started":"2022-07-07T15:40:03.580857Z","shell.execute_reply":"2022-07-07T15:40:03.595768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T15:40:45.200669Z","iopub.execute_input":"2022-07-07T15:40:45.201228Z","iopub.status.idle":"2022-07-07T15:40:45.212887Z","shell.execute_reply.started":"2022-07-07T15:40:45.201194Z","shell.execute_reply":"2022-07-07T15:40:45.211764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"custom_short_dict = {'AFAIK':'As Far As I Know',\n'AFK':'Away From Keyboard',\n'ASAP':'As Soon As Possible',\n'ATK':'At The Keyboard',\n'ATM':'At The Moment',\n'A3':'Anytime, Anywhere, Anyplace',\n'BAK':'Back At Keyboard',\n'BBL':'Be Back Later',\n'BBS':'Be Back Soon',\n'BFN':'Bye For Now',\n'B4N':'Bye For Now',\n'BRB':'Be Right Back',\n'BRT':'Be Right There',\n'BTW':'By The Way',\n'B4':'Before',\n'B4N':'Bye For Now',\n'CU':'See You',\n'CUL8R':'See You Later',\n'CYA':'See You',\n'FAQ':'Frequently Asked Questions',\n'FC':'Fingers Crossed',\n'FWIW':'For What It is Worth',\n'FYI':'For Your Information',\n'GAL':'Get A Life',\n'GG':'Good Game',\n'GN':'Good Night',\n'GMTA':'Great Minds Think Alike',\n'GR8':'Great!',\n'G9':'Genius',\n'IC':'I See',\n'ICQ':'I Seek you',\n'ILU':'I Love You',\n'IMHO':'In My Honest',\n'IMO':'In My Opinion',\n'IOW':'In Other Words',\n'IRL':'In Real Life',\n'KISS':'Keep It Simple, Stupid',\n'LDR':'Long Distance Relationship',\n'LMAO':'Laugh My Ass',\n'LOL':'Laughing Out Loud',\n'LTNS':'Long Time No See',\n'L8R':'Later',\n'MTE':'My Thoughts Exactly',\n'M8':'Mate',\n'NRN':'No Reply Necessary',\n'OIC':'Oh I See',\n'PITA':'Pain In The Ass',\n'PRT':'Party',\n'PRW':'Parents Are Watching',\n'ROFL':'Rolling On The Floor Laughing',\n'ROFLOL':'Rolling On The Floor Laughing Out Loud',\n'ROTFLMAO':'Rolling On The Floor Laughing My Ass',\n'SK8':'Skate',\n'STATS':'Your sex and age',\n'ASL':'Age, Sex, Location',\n'THX':'Thank You',\n'TTFN':'Ta-Ta For Now!',\n'TTYL':'Talk To You Later',\n'U':'You',\n'U2':'You Too',\n'U4E':'Yours For Ever',\n'WB':'Welcome Back',\n'WTF':'What The Fuck',\n'WTG':'Way To Go!',\n'WUF':'Where Are You From?',\n'W8':'Wait'}","metadata":{"execution":{"iopub.status.busy":"2022-07-07T15:40:49.195946Z","iopub.execute_input":"2022-07-07T15:40:49.196319Z","iopub.status.idle":"2022-07-07T15:40:49.209080Z","shell.execute_reply.started":"2022-07-07T15:40:49.196289Z","shell.execute_reply":"2022-07-07T15:40:49.208218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_data.replace(regex=custom_short_dict, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T15:40:53.425655Z","iopub.execute_input":"2022-07-07T15:40:53.426442Z","iopub.status.idle":"2022-07-07T15:40:55.367317Z","shell.execute_reply.started":"2022-07-07T15:40:53.426404Z","shell.execute_reply":"2022-07-07T15:40:55.366542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The Below code show some king of basic Preprocessing for Natural Language Preprocessing.**\n\nThe Preprocessing steps are as follows:\n> 1. Removing Stopwords.\n> 2. Removing Punctuations & Symbols.\n> 3. Removing Http Code using Beautiful Soup.","metadata":{}},{"cell_type":"code","source":"stopwords = [\"a\", \"about\", \"above\", \"after\", \"again\", \"against\", \"all\", \"am\", \"an\", \"and\", \"any\", \"are\", \"as\", \"at\",\n             \"be\", \"because\", \"been\", \"before\", \"being\", \"below\", \"between\", \"both\", \"but\", \"by\", \"could\", \"did\", \"do\",\n             \"does\", \"doing\", \"down\", \"during\", \"each\", \"few\", \"for\", \"from\", \"further\", \"had\", \"has\", \"have\", \"having\",\n             \"he\", \"hed\", \"hes\", \"her\", \"here\", \"heres\", \"hers\", \"herself\", \"him\", \"himself\", \"his\", \"how\",\n             \"hows\", \"i\", \"id\", \"ill\", \"im\", \"ive\", \"if\", \"in\", \"into\", \"is\", \"it\", \"its\", \"itself\",\n             \"lets\", \"me\", \"more\", \"most\", \"my\", \"myself\", \"nor\", \"of\", \"on\", \"once\", \"only\", \"or\", \"other\", \"ought\",\n             \"our\", \"ours\", \"ourselves\", \"out\", \"over\", \"own\", \"same\", \"she\", \"shed\", \"shell\", \"shes\", \"should\",\n             \"so\", \"some\", \"such\", \"than\", \"that\", \"thats\", \"the\", \"their\", \"theirs\", \"them\", \"themselves\", \"then\",\n             \"there\", \"theres\", \"these\", \"they\", \"theyd\", \"theyll\", \"theyre\", \"theyve\", \"this\", \"those\", \"through\",\n             \"to\", \"too\", \"under\", \"until\", \"up\", \"very\", \"was\", \"we\", \"wed\", \"well\", \"were\", \"weve\", \"were\",\n             \"what\", \"whats\", \"when\", \"whens\", \"where\", \"wheres\", \"which\", \"while\", \"who\", \"whos\", \"whom\", \"why\",\n             \"whys\", \"with\", \"would\", \"you\", \"youd\", \"youll\", \"youre\", \"youve\", \"your\", \"yours\", \"yourself\",\n             \"yourselves\"]\n\ntable = str.maketrans('', '', string.punctuation)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T15:41:29.673035Z","iopub.execute_input":"2022-07-07T15:41:29.673393Z","iopub.status.idle":"2022-07-07T15:41:29.685517Z","shell.execute_reply.started":"2022-07-07T15:41:29.673365Z","shell.execute_reply":"2022-07-07T15:41:29.684502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def sentence_replication(sentence):\n    sentence = re.sub(r'http\\S+', '', sentence)\n    sentence = sentence.replace(\",\", \" , \")\n    sentence = sentence.replace(\".\", \" . \")\n    sentence = sentence.replace(\"-\", \" - \")\n    sentence = sentence.replace(\"/\", \" / \")\n    sentence = sentence.replace(\"\\\\\", \" \\\\ \")\n    sentence = sentence.replace(\"0\", \" \")\n    sentence = sentence.replace(\"1\", \" \")\n    sentence = sentence.replace(\"2\", \" \")\n    sentence = sentence.replace(\"3\", \" \")\n    sentence = sentence.replace(\"4\", \" \")\n    sentence = sentence.replace(\"5\", \" \")\n    sentence = sentence.replace(\"6\", \" \")\n    sentence = sentence.replace(\"7\", \" \")\n    sentence = sentence.replace(\"8\", \" \")\n    sentence = sentence.replace(\"9\", \" \")\n    soup = BeautifulSoup(sentence)\n    sentence = soup.getText()\n    return sentence\n\ndef PreProcessing(df, training=True):\n    \n    df['keyword'] = df['keyword'].fillna('')\n    df['location'] = df['location'].fillna('')\n    \n    SENTENCE = []\n    LABELS = []\n    \n    for idx in df.index:\n        sentence = raw_data.iloc[idx, 1].lower()\n        sentence += ' '\n        sentence += raw_data.iloc[idx, 2].lower()\n        sentence += ' '\n        sentence += raw_data.iloc[idx, 3].lower()\n        sentence = sentence_replication(sentence)\n        words = sentence.split()\n        filtered_sentence = ''\n        for word in words:\n            word = word.translate(table)\n            if word not in stopwords:\n                filtered_sentence = filtered_sentence + word + ' '\n        SENTENCE.append(filtered_sentence)\n        \n        if training:\n            label = raw_data.iloc[idx, 4]\n            LABELS.append(label)\n            \n    if training:\n        return np.array(SENTENCE), np.array(LABELS)\n    \n    return np.array(SENTENCE)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T15:41:30.953147Z","iopub.execute_input":"2022-07-07T15:41:30.953547Z","iopub.status.idle":"2022-07-07T15:41:30.969835Z","shell.execute_reply.started":"2022-07-07T15:41:30.953495Z","shell.execute_reply":"2022-07-07T15:41:30.968874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X, Y = PreProcessing(raw_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T15:41:32.021082Z","iopub.execute_input":"2022-07-07T15:41:32.021963Z","iopub.status.idle":"2022-07-07T15:41:36.273712Z","shell.execute_reply.started":"2022-07-07T15:41:32.021917Z","shell.execute_reply":"2022-07-07T15:41:36.272856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model():\n    text_input = tf.keras.layers.Input(shape=(), dtype=tf.string, name='text_input')\n    preprocessed_text = bert_preprocessor(text_input)\n    output = bert_encoder(preprocessed_text)\n    o = tf.keras.layers.Dropout(0.1, name='Dropout')(output['pooled_output'])\n    o = tf.keras.layers.Dense(units=1, activation='sigmoid')(o)\n    \n    model = tf.keras.Model(inputs=[text_input], outputs=[o])\n    \n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-07-07T15:41:36.275230Z","iopub.execute_input":"2022-07-07T15:41:36.275636Z","iopub.status.idle":"2022-07-07T15:41:36.281630Z","shell.execute_reply.started":"2022-07-07T15:41:36.275608Z","shell.execute_reply":"2022-07-07T15:41:36.280873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> <h2 style='color:#ff6f00'> Model summary shows that we need to only train the Parameter of our DNN i.e. 768 weights + 1 bias = 769. </h2>","metadata":{}},{"cell_type":"code","source":"tweet_model = build_model()\ntweet_model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T15:41:37.121147Z","iopub.execute_input":"2022-07-07T15:41:37.121504Z","iopub.status.idle":"2022-07-07T15:41:37.439235Z","shell.execute_reply.started":"2022-07-07T15:41:37.121474Z","shell.execute_reply":"2022-07-07T15:41:37.438584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = tf.data.Dataset.from_tensor_slices((X, Y))\ntrain_dataset = train_dataset.shuffle(64).batch(BATCH_SIZE)\ntrain_dataset = train_dataset.cache()\ntrain_dataset = train_dataset.prefetch(AUTO)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T15:41:41.082289Z","iopub.execute_input":"2022-07-07T15:41:41.082793Z","iopub.status.idle":"2022-07-07T15:41:41.099355Z","shell.execute_reply.started":"2022-07-07T15:41:41.082751Z","shell.execute_reply":"2022-07-07T15:41:41.098601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_history = tweet_model.fit(train_dataset, epochs=20)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T15:41:42.028135Z","iopub.execute_input":"2022-07-07T15:41:42.028891Z","iopub.status.idle":"2022-07-07T15:43:12.926834Z","shell.execute_reply.started":"2022-07-07T15:41:42.028856Z","shell.execute_reply":"2022-07-07T15:43:12.925407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2 style='background:#FF6F00; color:white; padding:15px 0;'> Visualization of Model Training Accuracy & Loss </h2>","metadata":{}},{"cell_type":"markdown","source":"> **Unavailable**","metadata":{}},{"cell_type":"code","source":"#hist_df = pd.DataFrame(model_history.history)\n#fig = px.line(hist_df)\n#fig.update_layout(template='plotly_dark', width=1200, title='Metrics')\n#fig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T14:57:27.889236Z","iopub.execute_input":"2022-07-07T14:57:27.890966Z","iopub.status.idle":"2022-07-07T14:57:27.900567Z","shell.execute_reply.started":"2022-07-07T14:57:27.890872Z","shell.execute_reply":"2022-07-07T14:57:27.896678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2 style='background:#FF6F00; color:white; padding:15px 0;'> Test Submission </h2>","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv('../input/nlp-getting-started/test.csv')\ntest_df.replace(regex=custom_short_dict, inplace=True)\nid_col = test_df.iloc[:, 0]\nx_test = PreProcessing(test_df, training=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T15:43:21.660743Z","iopub.execute_input":"2022-07-07T15:43:21.661107Z","iopub.status.idle":"2022-07-07T15:43:24.109018Z","shell.execute_reply.started":"2022-07-07T15:43:21.661077Z","shell.execute_reply":"2022-07-07T15:43:24.108217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = tweet_model.predict(x_test)\npred = np.where(pred>=0.5, 1, 0)\npred = pred.reshape(-1)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T14:59:34.145955Z","iopub.execute_input":"2022-07-07T14:59:34.146596Z","iopub.status.idle":"2022-07-07T15:00:16.715005Z","shell.execute_reply.started":"2022-07-07T14:59:34.146548Z","shell.execute_reply":"2022-07-07T15:00:16.713586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = {'Id':id_col, 'target':pred}\ndf = pd.DataFrame(df)\ndf.to_csv('Submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style='background:#FF6F00;'>\n    <h1 style='padding:15px 0 0 0; color:white'> <center> Thanks for Reading 😀 </center></h1>\n</div>","metadata":{}},{"cell_type":"markdown","source":"<h2 style='padding:15px 0 0 0; background:#FF00D6; color:white'> <center> 🪧 Work in Progress 🪧 </center></h2>","metadata":{}}]}