{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-30T22:28:30.405457Z","iopub.execute_input":"2022-07-30T22:28:30.405886Z","iopub.status.idle":"2022-07-30T22:28:30.416350Z","shell.execute_reply.started":"2022-07-30T22:28:30.405846Z","shell.execute_reply":"2022-07-30T22:28:30.414955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Install dependencies","metadata":{}},{"cell_type":"code","source":"!pip install fuzz","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:28:30.523877Z","iopub.execute_input":"2022-07-30T22:28:30.524635Z","iopub.status.idle":"2022-07-30T22:28:45.875607Z","shell.execute_reply.started":"2022-07-30T22:28:30.524602Z","shell.execute_reply":"2022-07-30T22:28:45.874387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import the dataset","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:28:45.878074Z","iopub.execute_input":"2022-07-30T22:28:45.878385Z","iopub.status.idle":"2022-07-30T22:28:45.884602Z","shell.execute_reply.started":"2022-07-30T22:28:45.878353Z","shell.execute_reply":"2022-07-30T22:28:45.883372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = '/kaggle/input/'\npath = path+os.listdir(path)[0]+'/'\nprint(path)\nos.listdir(path)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:28:45.886153Z","iopub.execute_input":"2022-07-30T22:28:45.886653Z","iopub.status.idle":"2022-07-30T22:28:45.904153Z","shell.execute_reply.started":"2022-07-30T22:28:45.886622Z","shell.execute_reply":"2022-07-30T22:28:45.902636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\ntrain = pd.read_csv(path+'train.csv')\ntest = pd.read_csv(path+'test.csv')\nsubmission = pd.read_csv(path+'sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:28:45.907795Z","iopub.execute_input":"2022-07-30T22:28:45.909019Z","iopub.status.idle":"2022-07-30T22:28:46.073447Z","shell.execute_reply.started":"2022-07-30T22:28:45.908967Z","shell.execute_reply":"2022-07-30T22:28:46.072273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:28:46.075153Z","iopub.execute_input":"2022-07-30T22:28:46.075462Z","iopub.status.idle":"2022-07-30T22:28:46.096442Z","shell.execute_reply.started":"2022-07-30T22:28:46.075431Z","shell.execute_reply":"2022-07-30T22:28:46.095546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:28:46.097574Z","iopub.execute_input":"2022-07-30T22:28:46.098197Z","iopub.status.idle":"2022-07-30T22:28:46.105152Z","shell.execute_reply.started":"2022-07-30T22:28:46.098157Z","shell.execute_reply":"2022-07-30T22:28:46.104178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:28:46.106331Z","iopub.execute_input":"2022-07-30T22:28:46.107154Z","iopub.status.idle":"2022-07-30T22:28:46.131635Z","shell.execute_reply.started":"2022-07-30T22:28:46.107116Z","shell.execute_reply":"2022-07-30T22:28:46.130005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.dropna()\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:28:46.133821Z","iopub.execute_input":"2022-07-30T22:28:46.134147Z","iopub.status.idle":"2022-07-30T22:28:46.158481Z","shell.execute_reply.started":"2022-07-30T22:28:46.134108Z","shell.execute_reply":"2022-07-30T22:28:46.157569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def map_sentiment(sentiment):\n    if sentiment == 'neutral':\n        return 0\n    elif sentiment == 'negative':\n        return 1\n    else:\n        return 2","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:28:46.160075Z","iopub.execute_input":"2022-07-30T22:28:46.160720Z","iopub.status.idle":"2022-07-30T22:28:46.166122Z","shell.execute_reply.started":"2022-07-30T22:28:46.160681Z","shell.execute_reply":"2022-07-30T22:28:46.165260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train['sentiment'] = train['sentiment'].apply(lambda x: map_sentiment(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:28:46.170687Z","iopub.execute_input":"2022-07-30T22:28:46.171341Z","iopub.status.idle":"2022-07-30T22:28:46.176873Z","shell.execute_reply.started":"2022-07-30T22:28:46.171294Z","shell.execute_reply":"2022-07-30T22:28:46.175850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test['sentiment'] = test['sentiment'].apply(lambda x: map_sentiment(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:28:46.178689Z","iopub.execute_input":"2022-07-30T22:28:46.179468Z","iopub.status.idle":"2022-07-30T22:28:46.187947Z","shell.execute_reply.started":"2022-07-30T22:28:46.179420Z","shell.execute_reply":"2022-07-30T22:28:46.186723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def jaccard(str1, str2): \n    a = set(str1.lower().split()) \n    b = set(str2.lower().split())\n    c = a.intersection(b)\n    return float(len(c)) / (len(a) + len(b) - len(c))","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:28:46.189353Z","iopub.execute_input":"2022-07-30T22:28:46.190548Z","iopub.status.idle":"2022-07-30T22:28:46.200949Z","shell.execute_reply.started":"2022-07-30T22:28:46.190471Z","shell.execute_reply":"2022-07-30T22:28:46.199711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"markdown","source":"### Lower case","metadata":{}},{"cell_type":"code","source":"train['text']= train['text'].apply(lambda x: x.lower())\ntest['text']= test['text'].apply(lambda x: x.lower())\ntrain['selected_text']= train['selected_text'].apply(lambda x: x.lower())","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:28:46.202265Z","iopub.execute_input":"2022-07-30T22:28:46.202606Z","iopub.status.idle":"2022-07-30T22:28:46.242510Z","shell.execute_reply.started":"2022-07-30T22:28:46.202576Z","shell.execute_reply":"2022-07-30T22:28:46.241291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Removing Hyper-links","metadata":{}},{"cell_type":"code","source":"import re\n\ndef remove_hyperlinks(text):\n    hyperlinkfree=re.sub('https?://\\S+|www\\.\\S+', '', text)\n    return hyperlinkfree\ntrain['text']=train['text'].apply(lambda x:remove_hyperlinks(x))\ntest['text']=test['text'].apply(lambda x:remove_hyperlinks(x))\ntrain['selected_text']=train['selected_text'].apply(lambda x:remove_hyperlinks(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:28:46.244283Z","iopub.execute_input":"2022-07-30T22:28:46.244788Z","iopub.status.idle":"2022-07-30T22:28:46.432081Z","shell.execute_reply.started":"2022-07-30T22:28:46.244739Z","shell.execute_reply":"2022-07-30T22:28:46.430810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Removing Numbers, Angular Brackets, Square Brackets, ‘\\n’ character, replacing **** by <ABUSE>word","metadata":{}},{"cell_type":"code","source":"def remove(text):\n    text=re.sub('\\S*\\d\\S*',' ',text) #Removing Numbers\n    text=re.sub('<.*?>+',' ',text)   #Removing Angular Brackets\n    text=re.sub('\\[.*?\\]',' ',text)  #Removing Square Brackets\n    text=re.sub('\\n',' ',text)       #Removing '\\n' character \n    text=re.sub('\\*+','<ABUSE>',text) #Replacing **** by ABUSE word\n    return text\ntrain['text']=train['text'].apply(lambda x:remove(x))\ntest['text']=test['text'].apply(lambda x:remove(x))\ntrain['selected_text']=train['selected_text'].apply(lambda x:remove(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:28:46.433848Z","iopub.execute_input":"2022-07-30T22:28:46.434332Z","iopub.status.idle":"2022-07-30T22:28:47.469597Z","shell.execute_reply.started":"2022-07-30T22:28:46.434286Z","shell.execute_reply":"2022-07-30T22:28:47.468208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Remove punctiation","metadata":{}},{"cell_type":"code","source":"import string\n\ndef remove_punctuation(text):\n    punctuationfree=\"\".join([i for i in text if i not in string.punctuation])\n    return punctuationfree\n\ntrain['text']=train['text'].apply(lambda x:remove_punctuation(x))\ntest['text']=test['text'].apply(lambda x:remove_punctuation(x))\ntrain['selected_text']=train['selected_text'].apply(lambda x:remove_punctuation(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:28:47.471428Z","iopub.execute_input":"2022-07-30T22:28:47.472145Z","iopub.status.idle":"2022-07-30T22:28:47.911999Z","shell.execute_reply.started":"2022-07-30T22:28:47.472092Z","shell.execute_reply":"2022-07-30T22:28:47.910774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Spelling correction","metadata":{}},{"cell_type":"code","source":"def wrong_words(text,selected):\n    words=[]\n    text=text.split()\n    selected=selected.split()\n    for i in selected:\n        if i not in text:\n            words.append(i)\n    if len(words)>0:\n        return \" \".join(words)\n    else:\n        return '++++'\n    \ntrain['spelling']=train.apply(lambda x: wrong_words(x.text,x.selected_text),axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:28:47.913811Z","iopub.execute_input":"2022-07-30T22:28:47.914292Z","iopub.status.idle":"2022-07-30T22:28:48.876985Z","shell.execute_reply.started":"2022-07-30T22:28:47.914245Z","shell.execute_reply":"2022-07-30T22:28:48.875780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_text(x):\n    selected=x[0]\n    spelling=x[1]\n    selected=selected.split()\n    selected.remove(spelling) \n    return \" \".join(selected)\n\ntrain['selected_text']=train[['selected_text','spelling']].apply(lambda x: remove_text(x) if len(x['spelling'])==1  else x['selected_text'],axis=1)\ntrain['spelling']=train.apply(lambda x: wrong_words(x.text,x.selected_text),axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:28:48.878923Z","iopub.execute_input":"2022-07-30T22:28:48.879569Z","iopub.status.idle":"2022-07-30T22:28:50.406110Z","shell.execute_reply.started":"2022-07-30T22:28:48.879515Z","shell.execute_reply":"2022-07-30T22:28:50.404941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from fuzzywuzzy import fuzz\nfrom fuzzywuzzy import process\n\ndef matching(x):\n    text=x[0]\n    selected=x[1]\n    spelling=x[2]\n    text=text.split()\n    selected=selected.split()\n    spelling=spelling.split()\n    for s in spelling:\n        for t in text:\n            if s in selected:\n                if(fuzz.ratio(t,s)>55): \n                    index=selected.index(s)\n                    selected[index]=t\n    return \" \".join(selected)\n\ntrain['selected_text']=train[['text','selected_text','spelling']].apply(lambda x: matching(x) if x['spelling']!='++++'  else x['selected_text'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:28:50.407541Z","iopub.execute_input":"2022-07-30T22:28:50.407849Z","iopub.status.idle":"2022-07-30T22:28:51.038809Z","shell.execute_reply.started":"2022-07-30T22:28:50.407819Z","shell.execute_reply":"2022-07-30T22:28:51.037281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.loc[train['spelling']=='++++']","metadata":{"execution":{"iopub.status.busy":"2022-07-31T01:54:58.864645Z","iopub.execute_input":"2022-07-31T01:54:58.865087Z","iopub.status.idle":"2022-07-31T01:54:58.880474Z","shell.execute_reply.started":"2022-07-31T01:54:58.865053Z","shell.execute_reply":"2022-07-31T01:54:58.879282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Deep Learning Models","metadata":{}},{"cell_type":"markdown","source":"### RNN","metadata":{}},{"cell_type":"code","source":"from keras.preprocessing import text, sequence\nfrom keras.preprocessing.text import Tokenizer","metadata":{"execution":{"iopub.status.busy":"2022-07-30T23:55:03.278191Z","iopub.execute_input":"2022-07-30T23:55:03.278651Z","iopub.status.idle":"2022-07-30T23:55:03.284402Z","shell.execute_reply.started":"2022-07-30T23:55:03.278617Z","shell.execute_reply":"2022-07-30T23:55:03.283369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_valid, _, _ = train_test_split(train, train, test_size=0.3, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T01:55:05.556957Z","iopub.execute_input":"2022-07-31T01:55:05.557671Z","iopub.status.idle":"2022-07-31T01:55:05.576275Z","shell.execute_reply.started":"2022-07-31T01:55:05.557636Z","shell.execute_reply":"2022-07-31T01:55:05.575193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_text = X_train['text'].values\nvalid_text = X_valid['text'].values\n\ntrain_sentiment = X_train['sentiment'].values\nvalid_sentiment = X_valid['sentiment'].values","metadata":{"execution":{"iopub.status.busy":"2022-07-31T01:55:10.265889Z","iopub.execute_input":"2022-07-31T01:55:10.266707Z","iopub.status.idle":"2022-07-31T01:55:10.276015Z","shell.execute_reply.started":"2022-07-31T01:55:10.266658Z","shell.execute_reply":"2022-07-31T01:55:10.274356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# using keras tokenizer\ntoken1=text.Tokenizer(num_words=None)\nmax_len_text=32\n\ntoken1.fit_on_texts(list(train_text))\ntrain_text=token1.texts_to_sequences(train_text)\nvalid_text=token1.texts_to_sequences(valid_text)\n\n#zero pad the sequences\ntrain_text=sequence.pad_sequences(train_text,maxlen=max_len_text,padding='post')\nvalid_text=sequence.pad_sequences(valid_text,maxlen=max_len_text,padding='post')\nword_index_text=token1.word_index\n\n\n# using keras tokenizer\ntoken2=text.Tokenizer(num_words=None)\nmax_len_sentiment=1\n\ntoken2.fit_on_texts(list(train_sentiment))\ntrain_sentiment=token2.texts_to_sequences(train_sentiment)\nvalid_sentiment=token2.texts_to_sequences(valid_sentiment)\n\n#zero pad the sequences\ntrain_sentiment=sequence.pad_sequences(train_sentiment,maxlen=max_len_sentiment,padding='post')\nvalid_sentiment=sequence.pad_sequences(valid_sentiment,maxlen=max_len_sentiment,padding='post')\nword_index_sentiment=token2.word_index","metadata":{"execution":{"iopub.status.busy":"2022-07-31T01:55:13.317697Z","iopub.execute_input":"2022-07-31T01:55:13.318761Z","iopub.status.idle":"2022-07-31T01:55:14.719642Z","shell.execute_reply.started":"2022-07-31T01:55:13.318709Z","shell.execute_reply":"2022-07-31T01:55:14.718245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Preparing the embedding layer","metadata":{}},{"cell_type":"code","source":"!wget https://huggingface.co/stanfordnlp/glove/resolve/main/glove.840B.300d.zip","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:42:38.282040Z","iopub.execute_input":"2022-07-30T22:42:38.282414Z","iopub.status.idle":"2022-07-30T22:44:09.940630Z","shell.execute_reply.started":"2022-07-30T22:42:38.282384Z","shell.execute_reply":"2022-07-30T22:44:09.938708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!unzip glove*.zip","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:47:01.441605Z","iopub.execute_input":"2022-07-30T22:47:01.441989Z","iopub.status.idle":"2022-07-30T22:48:02.667568Z","shell.execute_reply.started":"2022-07-30T22:47:01.441957Z","shell.execute_reply":"2022-07-30T22:48:02.665970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls\n!pwd","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:48:35.461327Z","iopub.execute_input":"2022-07-30T22:48:35.461900Z","iopub.status.idle":"2022-07-30T22:48:37.031359Z","shell.execute_reply.started":"2022-07-30T22:48:35.461865Z","shell.execute_reply":"2022-07-30T22:48:37.029581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"glob_path = './glove.840B.300d.txt'","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:48:57.721235Z","iopub.execute_input":"2022-07-30T22:48:57.721716Z","iopub.status.idle":"2022-07-30T22:48:57.727567Z","shell.execute_reply.started":"2022-07-30T22:48:57.721679Z","shell.execute_reply":"2022-07-30T22:48:57.726088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\n\n# load the GloVe vectors in a dictionary:\nembeddings_index = {}\nwith open(glob_path, encoding='utf-8') as f:\n    for line in tqdm(f):\n        values = line.split(' ')\n        word = values[0]\n        coefs = np.asarray([float(val) for val in values[1:]])\n        embeddings_index[word] = coefs\nprint('Found %s word vectors.' % len(embeddings_index))","metadata":{"execution":{"iopub.status.busy":"2022-07-31T01:55:47.978285Z","iopub.execute_input":"2022-07-31T01:55:47.978734Z","iopub.status.idle":"2022-07-31T02:01:25.126271Z","shell.execute_reply.started":"2022-07-31T01:55:47.978700Z","shell.execute_reply":"2022-07-31T02:01:25.125082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create an embedding matrix for the words we have in the dataset\nembedding_matrix_text=np.zeros((len(word_index_text) + 1, 300))\nfor word, i in tqdm(word_index_text.items()):\n    embedding_vector=embeddings_index.get(word)\n    if embedding_vector is not None:\n        embedding_matrix_text[i]=embedding_vector\n\n# create an embedding matrix for the sentiments we have in the dataset\nembedding_matrix_sentiment=np.zeros((len(word_index_sentiment) + 1, 300))\nfor word, i in tqdm(word_index_sentiment.items()):\n    embedding_vector=embeddings_index.get(word)\n    if embedding_vector is not None:\n        embedding_matrix_sentiment[i]=embedding_vector","metadata":{"execution":{"iopub.status.busy":"2022-07-31T02:07:53.165505Z","iopub.execute_input":"2022-07-31T02:07:53.166203Z","iopub.status.idle":"2022-07-31T02:07:53.335050Z","shell.execute_reply.started":"2022-07-31T02:07:53.166164Z","shell.execute_reply":"2022-07-31T02:07:53.333693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LSTM model","metadata":{}},{"cell_type":"code","source":"from keras.layers import Input\nfrom keras.models import Sequential, Model\nfrom keras.layers import Dense, Dropout, Activation, Embedding, LSTM, BatchNormalization\nfrom keras.layers.merge import Concatenate\nfrom keras import regularizers\nfrom tensorflow.keras.optimizers import Adam","metadata":{"execution":{"iopub.status.busy":"2022-07-30T23:36:01.990084Z","iopub.execute_input":"2022-07-30T23:36:01.990457Z","iopub.status.idle":"2022-07-30T23:36:01.998231Z","shell.execute_reply.started":"2022-07-30T23:36:01.990425Z","shell.execute_reply":"2022-07-30T23:36:01.996708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_input=Input(shape=(max_len_text,),name='text_input')\nembd_text=Embedding(len(word_index_text)+1, #embedding layer with glove vectors as embeddings\n                    300,\n                    weights=[embedding_matrix_text],\n                    input_length=max_len_text,\n                    trainable=False,mask_zero=True,name='embedding_text')(text_input) #masking the input values with mask_zero= True\n\nsentiment_input=Input(shape=(max_len_sentiment,),name='sentiment_input')\nembd_sentiment=Embedding(len(word_index_sentiment)+1, #embedding layer with glove vectors as embeddings\n                    300,\n                    weights=[embedding_matrix_sentiment],\n                    input_length=max_len_text,\n                    trainable=False,mask_zero=True,name='embedding_sentiment')(sentiment_input) #masking the input values with mask_zero= True\n\ncon=Concatenate(axis=1)([embd_text,embd_sentiment])\nlstm=LSTM(64,return_sequences=True,kernel_regularizer=regularizers.l2(0.001),name='LSTM')(con) #lstm\n\n#dense layers with drop outs and batch normalisation\nm=Dense(32,activation=\"relu\",kernel_initializer=\"he_normal\",kernel_regularizer=regularizers.l2(0.001))(lstm) \nm=Dropout(0.5)(m)\nm=BatchNormalization()(m)\nm=Dense(4,activation=\"relu\", kernel_initializer=\"he_normal\",kernel_regularizer=regularizers.l2(0.001))(m)\noutput=Dense(1,activation='softmax',name='output')(m)\n\nmodel=Model(inputs=[text_input,sentiment_input],outputs=[output])","metadata":{"execution":{"iopub.status.busy":"2022-07-31T02:48:08.081982Z","iopub.execute_input":"2022-07-31T02:48:08.082508Z","iopub.status.idle":"2022-07-31T02:48:10.923866Z","shell.execute_reply.started":"2022-07-31T02:48:08.082457Z","shell.execute_reply":"2022-07-31T02:48:10.922387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_text\n#valid_text\n\n#train_sentiment\n#valid_sentiment","metadata":{"execution":{"iopub.status.busy":"2022-07-30T23:40:33.371730Z","iopub.execute_input":"2022-07-30T23:40:33.372220Z","iopub.status.idle":"2022-07-30T23:40:33.377751Z","shell.execute_reply.started":"2022-07-30T23:40:33.372185Z","shell.execute_reply":"2022-07-30T23:40:33.376703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_selected_text = X_train['selected_text'].values\nvalid_selected_text = X_valid['selected_text'].values\n\ntrain_selected_text=token1.texts_to_sequences(train_selected_text)\nvalid_selected_text=token1.texts_to_sequences(valid_selected_text)\n\ntrain_selected_text = np.array(train_selected_text)\nvalid_selected_text = np.array(valid_selected_text)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T02:08:05.380832Z","iopub.execute_input":"2022-07-31T02:08:05.381589Z","iopub.status.idle":"2022-07-31T02:08:05.726889Z","shell.execute_reply.started":"2022-07-31T02:08:05.381550Z","shell.execute_reply":"2022-07-31T02:08:05.725660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_output = []\nfor i in tqdm(range(train_text.shape[0])):\n    a = train_text[i]\n    sub = train_selected_text[i]\n    if len(sub)==0:\n        b = np.zeros((33,), dtype='int32')\n    else:\n        start = np.where(a == sub[0])[0][0]\n        end = start + len(sub)-1\n   \n        b = np.zeros((33,), dtype='int32')\n        c = np.ones((33,), dtype='int32')\n        b[start:end+1] = c[start:end+1]\n    output = b\n    train_output.append(output)\ntrain_output = np.array(train_output)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T02:44:32.868292Z","iopub.execute_input":"2022-07-31T02:44:32.868704Z","iopub.status.idle":"2022-07-31T02:44:33.302155Z","shell.execute_reply.started":"2022-07-31T02:44:32.868671Z","shell.execute_reply":"2022-07-31T02:44:33.301039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_output = []\n\nfor i in tqdm(range(valid_text.shape[0])):\n    a = valid_text[i]\n    sub = valid_selected_text[i]\n    if len(sub)==0:\n        b = np.zeros((33,), dtype='int32')\n    else:\n        start = np.where(a == sub[0])[0][0]\n        end = start + len(sub)-1\n   \n        b = np.zeros((33,), dtype='int32')\n        c = np.ones((33,), dtype='int32')\n        b[start:end+1] = c[start:end+1]\n    output = b\n    valid_output.append(output)\nvalid_output = np.array(valid_output)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T02:44:38.510982Z","iopub.execute_input":"2022-07-31T02:44:38.511403Z","iopub.status.idle":"2022-07-31T02:44:38.677776Z","shell.execute_reply.started":"2022-07-31T02:44:38.511371Z","shell.execute_reply":"2022-07-31T02:44:38.675233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T02:29:51.392288Z","iopub.execute_input":"2022-07-31T02:29:51.392839Z","iopub.status.idle":"2022-07-31T02:29:51.402295Z","shell.execute_reply.started":"2022-07-31T02:29:51.392795Z","shell.execute_reply":"2022-07-31T02:29:51.400807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow\ntensorflow.keras.utils.plot_model(model)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T00:30:09.375079Z","iopub.execute_input":"2022-07-31T00:30:09.375587Z","iopub.status.idle":"2022-07-31T00:30:11.533889Z","shell.execute_reply.started":"2022-07-31T00:30:09.375550Z","shell.execute_reply":"2022-07-31T00:30:11.532504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer='Adam',loss=\"categorical_crossentropy\",metrics=[\"accuracy\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-31T02:46:34.395999Z","iopub.execute_input":"2022-07-31T02:46:34.396405Z","iopub.status.idle":"2022-07-31T02:46:34.410841Z","shell.execute_reply.started":"2022-07-31T02:46:34.396372Z","shell.execute_reply":"2022-07-31T02:46:34.409394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_data = (train_text,train_sentiment)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T02:44:53.960871Z","iopub.execute_input":"2022-07-31T02:44:53.961316Z","iopub.status.idle":"2022-07-31T02:44:53.966572Z","shell.execute_reply.started":"2022-07-31T02:44:53.961283Z","shell.execute_reply":"2022-07-31T02:44:53.965396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val = (valid_text,valid_sentiment)\nval_data = (val,valid_output)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T02:45:05.561082Z","iopub.execute_input":"2022-07-31T02:45:05.561473Z","iopub.status.idle":"2022-07-31T02:45:05.567139Z","shell.execute_reply.started":"2022-07-31T02:45:05.561443Z","shell.execute_reply":"2022-07-31T02:45:05.565852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit([train_text, train_sentiment],train_output,epochs=20) # \"\"\",batch_size=128,validation_data=val_data,validation_batch_size=64\"\"\")","metadata":{"execution":{"iopub.status.busy":"2022-07-31T02:46:41.402259Z","iopub.execute_input":"2022-07-31T02:46:41.402702Z","iopub.status.idle":"2022-07-31T02:46:42.844258Z","shell.execute_reply.started":"2022-07-31T02:46:41.402668Z","shell.execute_reply":"2022-07-31T02:46:42.842702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def evaluate(model, tweet,sentiment):\n    tweet=str(tweet)\n    sentiment=str(sentiment)\n    tweet=tweet.lower()\n    #storing the puntuation free text\n    tweet=re.sub('https?://\\S+|www\\.\\S+', '', tweet)\n    tweet=re.sub('\\S*\\d\\S*',' ',tweet) #Removing Numbers\n    tweet=re.sub('<.*?>+',' ',tweet)   #Removing Angular Brackets\n    tweet=re.sub('\\[.*?\\]',' ',tweet)  #Removing Square Brackets\n    tweet=re.sub('\\n',' ',tweet)       #Removing '\\n' character \n    tweet=re.sub('\\*+','<ABUSE>',tweet) #Replacing **** by ABUSE word\n    tweet=\"\".join([i for i in tweet if i not in string.punctuation])\n    print(tweet,sentiment)\n\n    train_text=X_train['text'].values\n    train_sentiment=X_train['sentiment'].values\n\n    tokenizer1=Tokenizer(lower=True,split=' ',filters='!\"#$%&()*+,-./:;<=>?@[\\\\]^_{|}~\\t\\n',oov_token='<unw>')\n    tokenizer1.fit_on_texts(train_text)\n    max_len_text=32\n    train_text=tokenizer1.texts_to_sequences(train_text)\n    train_text=sequence.pad_sequences(train_text,maxlen=max_len_text,padding='post')\n    word_index_text=tokenizer1.word_index\n\n    tokenizer2=Tokenizer(lower=True,split=' ',filters='!\"#$%&()*+,-./:;<=>?@[\\\\]^_{|}~\\t\\n',oov_token='<unw>')\n    tokenizer2.fit_on_texts(train_sentiment)\n    max_len_sentiment=1\n    train_sentiment=tokenizer2.texts_to_sequences(train_sentiment)\n    train_sentiment=sequence.pad_sequences(train_sentiment,maxlen=max_len_sentiment,padding='post')\n    word_index_sentiment=tokenizer2.word_index\n \n    text=[]\n    for k in tweet.split():\n        if k in word_index_text:\n            text.append(word_index_text[k])\n        else:\n            text.append(0)\n    while len(text)<32:\n        text.append(0)\n    text=np.reshape(text,(1,32))\n    \n    s=1\n    if sentiment in word_index_sentiment:\n        s=word_index_sentiment[sentiment]\n        s=np.reshape(s,(1,1))\n\n    tweet_pred=model.predict([text,s])\n    tweet_pred=np.squeeze(tweet_pred)\n    tweet_pred=np.round(tweet_pred)\n    tweet_pred=np.reshape(tweet_pred,(1,33))\n    print(tweet_pred)\n\n    pred=[]\n    for vector in tweet_pred:\n        index=[]\n        for i,value in enumerate(vector):\n            if value == 1:\n                index.append(i)\n        pred.append(index)\n    pred=pred[0]\n    print(pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T00:33:29.488785Z","iopub.execute_input":"2022-07-31T00:33:29.489261Z","iopub.status.idle":"2022-07-31T00:33:29.510133Z","shell.execute_reply.started":"2022-07-31T00:33:29.489225Z","shell.execute_reply":"2022-07-31T00:33:29.508554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluate(model,'Sooo sad', 'negative')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T00:33:34.517820Z","iopub.execute_input":"2022-07-31T00:33:34.518296Z","iopub.status.idle":"2022-07-31T00:33:35.861173Z","shell.execute_reply.started":"2022-07-31T00:33:34.518261Z","shell.execute_reply":"2022-07-31T00:33:35.859780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Bi-LSTM model","metadata":{}},{"cell_type":"code","source":"text_input=Input(shape=(max_len_text,),name='text_input')\nembd_text=Embedding(len(word_index_text)+1, #embedding layer with glove vectors as embeddings\n                    300,\n                    weights=[embedding_matrix_text],\n                    input_length=max_len_text,\n                    trainable=False,mask_zero=True,name='embedding_text')(text_input) #masking the input values with mask_zero= True\n\nsentiment_input=Input(shape=(max_len_sentiment,),name='sentiment_input')\nembd_sentiment=Embedding(len(word_index_sentiment)+1, #embedding layer with glove vectors as embeddings\n                    300,\n                    weights=[embedding_matrix_sentiment],\n                    input_length=max_len_text,\n                    trainable=False,mask_zero=True,name='embedding_sentiment')(sentiment_input) #masking the input values with mask_zero= True\n\ncon=Concatenate(axis=1)([embd_text,embd_sentiment])\nlstm=Bidirectional(LSTM(16,return_sequences=True,dropout=0.4,name='lstm'))(con)\n\n#dense layers with drop outs and batch normalisation\nm=Dense(8,activation=\"relu\",kernel_regularizer=regularizers.l2(0.0001))(lstm) \nm=Dropout(0.5)(m)\nm=Dense(4,activation=\"relu\",kernel_regularizer=regularizers.l2(0.0001))(m)\noutput=Dense(1,activation='sigmoid',name='output')(m)\n\nmodel=Model(inputs=[text_input,sentiment_input],outputs=[output])","metadata":{},"execution_count":null,"outputs":[]}]}