{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2022-07-12T01:14:19.198159Z","iopub.execute_input":"2022-07-12T01:14:19.198645Z","iopub.status.idle":"2022-07-12T01:14:19.230393Z","shell.execute_reply.started":"2022-07-12T01:14:19.198552Z","shell.execute_reply":"2022-07-12T01:14:19.229015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Dataset","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom IPython.display import display_html \n\ntrainDf = pd.read_csv('../input/nlp-getting-started/train.csv')\ntestDf = pd.read_csv('../input/nlp-getting-started/test.csv')\n\ntrainDf.shape, testDf.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:39.735388Z","iopub.execute_input":"2022-07-12T01:14:39.735809Z","iopub.status.idle":"2022-07-12T01:14:39.811394Z","shell.execute_reply.started":"2022-07-12T01:14:39.735775Z","shell.execute_reply":"2022-07-12T01:14:39.810411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(trainDf.head(3))\ndisplay(testDf.head(3))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:43.099677Z","iopub.execute_input":"2022-07-12T01:14:43.100088Z","iopub.status.idle":"2022-07-12T01:14:43.127475Z","shell.execute_reply.started":"2022-07-12T01:14:43.100055Z","shell.execute_reply":"2022-07-12T01:14:43.126533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainDf.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:43.547941Z","iopub.execute_input":"2022-07-12T01:14:43.549042Z","iopub.status.idle":"2022-07-12T01:14:43.585135Z","shell.execute_reply.started":"2022-07-12T01:14:43.549002Z","shell.execute_reply":"2022-07-12T01:14:43.583981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"markdown","source":"- Check NaNs","metadata":{}},{"cell_type":"code","source":"import plotly.graph_objects as go\n\nfig = go.Figure([go.Bar(x=trainDf.columns, y=np.sum(trainDf.isna().values, axis=0))])\nfig.update_traces(marker_color='rgb(158,202,225)', marker_line_color='rgb(8,48,107)',\n                  marker_line_width=1.5, opacity=0.6)\nfig.update_layout(title_text='Missing values')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:44.113741Z","iopub.execute_input":"2022-07-12T01:14:44.114252Z","iopub.status.idle":"2022-07-12T01:14:44.295051Z","shell.execute_reply.started":"2022-07-12T01:14:44.114188Z","shell.execute_reply":"2022-07-12T01:14:44.293927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Check Duplicates","metadata":{}},{"cell_type":"code","source":"trainDf.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:44.773920Z","iopub.execute_input":"2022-07-12T01:14:44.774318Z","iopub.status.idle":"2022-07-12T01:14:44.793284Z","shell.execute_reply.started":"2022-07-12T01:14:44.774284Z","shell.execute_reply":"2022-07-12T01:14:44.792079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Check Dataset Balancing","metadata":{}},{"cell_type":"code","source":"from plotly.subplots import make_subplots\nimport plotly.graph_objects as go\n\n\nfig = make_subplots(rows=1, cols=2,specs=[[{\"type\": \"domain\"}, {\"type\": \"xy\"}]])\n\nfig.add_trace(\n    go.Pie(labels=['Not Disaster', 'Disaster'], values=trainDf.target.value_counts()),\n    row=1, col=1\n)\n\nfig.add_trace(\n    go.Bar(x=['Not Disaster', 'Disaster'], y=trainDf.target.value_counts(),marker_color=['orange', 'green']),\n    row=1, col=2\n)\n\n\n\nfig.update_layout(title_text=\"Target Distribution in Dataset\" ,showlegend=False)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:44.839793Z","iopub.execute_input":"2022-07-12T01:14:44.840719Z","iopub.status.idle":"2022-07-12T01:14:45.055749Z","shell.execute_reply.started":"2022-07-12T01:14:44.840683Z","shell.execute_reply":"2022-07-12T01:14:45.054377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Check tweets distribution","metadata":{}},{"cell_type":"code","source":"#real disaster (1) or not (0)\n\nfig = go.Figure([go.Bar(x=trainDf[trainDf.target == 0].location.value_counts()[:20].index,\n                        y=trainDf[trainDf.target == 0].location.value_counts()[:20].values,name='not disaster'),\n                go.Bar(x=trainDf[trainDf.target == 1].location.value_counts()[:20].index,\n                        y=trainDf[trainDf.target == 1].location.value_counts()[:20].values,name='disaster')])\n\n\nfig.update_layout(title_text='Tweets distribution for each country',barmode='stack')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:45.057886Z","iopub.execute_input":"2022-07-12T01:14:45.058520Z","iopub.status.idle":"2022-07-12T01:14:45.091604Z","shell.execute_reply.started":"2022-07-12T01:14:45.058474Z","shell.execute_reply":"2022-07-12T01:14:45.090535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- show samples of data texts to find out required preprocessing steps","metadata":{}},{"cell_type":"code","source":"disasterDF = trainDf[trainDf.target==1] \nnotDisasterDf = trainDf[trainDf.target==0] \n\nprint('***** DisasterDF(1) ***** \\n')\nprint(disasterDF.text.iloc[13])\nprint(disasterDF.text.iloc[5])\nprint(disasterDF.text.iloc[34])\nprint(disasterDF.text.iloc[4])\nprint(disasterDF.text.iloc[1])\n\nprint('\\n***** Not Disaster(0) ***** \\n')\nprint(notDisasterDf.text.iloc[2])\nprint(notDisasterDf.text.iloc[14])\nprint(notDisasterDf.text.iloc[478])\nprint(notDisasterDf.text.iloc[7])\nprint(notDisasterDf.text.iloc[88])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:45.092630Z","iopub.execute_input":"2022-07-12T01:14:45.093475Z","iopub.status.idle":"2022-07-12T01:14:45.105472Z","shell.execute_reply.started":"2022-07-12T01:14:45.093440Z","shell.execute_reply":"2022-07-12T01:14:45.104474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Cleaning and Preprocessing are:\n    - 1 Remove @word from all tweets\n    - 2 Remove numbers from all tweets\n    - 3 Remove hashtags(#) only from the tweets\n    - 4 Remove unchars word\n    - 5 Check for emails\n    - 6 Check for websites ( http...)\n    - 7 Normalizing","metadata":{}},{"cell_type":"markdown","source":"- Remove @word from all tweets","metadata":{}},{"cell_type":"code","source":"import re\n\ndfTrainModified = trainDf.copy()\ndfTrainModified['text']=dfTrainModified['text'].apply(lambda x: re.sub('(\\s*)@\\w+(\\s*)','', x))\n\ndfTrainModified.text[43], trainDf.text[43]","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:45.107043Z","iopub.execute_input":"2022-07-12T01:14:45.107849Z","iopub.status.idle":"2022-07-12T01:14:45.168331Z","shell.execute_reply.started":"2022-07-12T01:14:45.107816Z","shell.execute_reply":"2022-07-12T01:14:45.167325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Remove numbers from all tweets","metadata":{}},{"cell_type":"code","source":"dfTrainModified['text']=dfTrainModified['text'].apply(lambda x: re.sub('\\d+','', x))\ndfTrainModified.text[3], trainDf.text[3]","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:45.169777Z","iopub.execute_input":"2022-07-12T01:14:45.170324Z","iopub.status.idle":"2022-07-12T01:14:45.208285Z","shell.execute_reply.started":"2022-07-12T01:14:45.170293Z","shell.execute_reply":"2022-07-12T01:14:45.207247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Remove hashtags(#) only from the tweets","metadata":{}},{"cell_type":"code","source":"dfTrainModified['text']=dfTrainModified['text'].apply(lambda x: re.sub('#','', x))\ndfTrainModified.text[5], trainDf.text[5]","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:45.209804Z","iopub.execute_input":"2022-07-12T01:14:45.210344Z","iopub.status.idle":"2022-07-12T01:14:45.230272Z","shell.execute_reply.started":"2022-07-12T01:14:45.210312Z","shell.execute_reply":"2022-07-12T01:14:45.229090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Check for websites ( http...)","metadata":{}},{"cell_type":"code","source":"dfTrainModified['text'] = dfTrainModified['text'].apply(lambda x: re.sub('https?://\\S+|www\\.\\S+','',x))\ndfTrainModified.text[100], trainDf.text[100]","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:45.263683Z","iopub.execute_input":"2022-07-12T01:14:45.264057Z","iopub.status.idle":"2022-07-12T01:14:45.290140Z","shell.execute_reply.started":"2022-07-12T01:14:45.264027Z","shell.execute_reply":"2022-07-12T01:14:45.289089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Remove unchars word","metadata":{}},{"cell_type":"code","source":"dfTrainModified['text'] = dfTrainModified['text'].apply(lambda x: re.sub('[^A-Za-z]',' ',x))\ndfTrainModified.text[5], trainDf.text[5]","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:45.349471Z","iopub.execute_input":"2022-07-12T01:14:45.350488Z","iopub.status.idle":"2022-07-12T01:14:45.404451Z","shell.execute_reply.started":"2022-07-12T01:14:45.350448Z","shell.execute_reply":"2022-07-12T01:14:45.403496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Check for emails","metadata":{}},{"cell_type":"code","source":"result=0\nfor i in range(len(dfTrainModified)):\n    if re.findall(r'\\S+@(\\S+)', dfTrainModified.text[i]):\n        result +=1\nprint(result)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:45.431353Z","iopub.execute_input":"2022-07-12T01:14:45.431917Z","iopub.status.idle":"2022-07-12T01:14:45.541002Z","shell.execute_reply.started":"2022-07-12T01:14:45.431886Z","shell.execute_reply":"2022-07-12T01:14:45.539755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Normalizing","metadata":{}},{"cell_type":"markdown","source":"- Lower all words","metadata":{}},{"cell_type":"code","source":"dfTrainModified['text']=dfTrainModified['text'].apply(lambda x: x.lower())\ndfTrainModified.text[5], trainDf.text[5]","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:45.584561Z","iopub.execute_input":"2022-07-12T01:14:45.585058Z","iopub.status.idle":"2022-07-12T01:14:45.600524Z","shell.execute_reply.started":"2022-07-12T01:14:45.585012Z","shell.execute_reply":"2022-07-12T01:14:45.599565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Remove stopwords","metadata":{}},{"cell_type":"code","source":"import nltk \nfrom nltk.corpus import stopwords\n#nltk.download('stopwords')\nstop_words=stopwords.words('english')\n\ndfTrainModified['text']=dfTrainModified['text'].apply(lambda x : [word for word in x.split()  if word not in stop_words])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:45.602324Z","iopub.execute_input":"2022-07-12T01:14:45.602922Z","iopub.status.idle":"2022-07-12T01:14:46.987434Z","shell.execute_reply.started":"2022-07-12T01:14:45.602888Z","shell.execute_reply":"2022-07-12T01:14:46.986146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfTrainModified.text[98], trainDf.text[98]","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:46.989731Z","iopub.execute_input":"2022-07-12T01:14:46.990932Z","iopub.status.idle":"2022-07-12T01:14:47.000580Z","shell.execute_reply.started":"2022-07-12T01:14:46.990884Z","shell.execute_reply":"2022-07-12T01:14:46.999423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Lemmatization","metadata":{}},{"cell_type":"code","source":"from nltk.corpus import wordnet\nfrom nltk.stem import WordNetLemmatizer\n\nlemmatizer = WordNetLemmatizer()\n\ndfTrainModified['text']=dfTrainModified['text'].apply(lambda x: [lemmatizer.lemmatize(word) for word in x])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:47.002447Z","iopub.execute_input":"2022-07-12T01:14:47.002815Z","iopub.status.idle":"2022-07-12T01:14:49.011742Z","shell.execute_reply.started":"2022-07-12T01:14:47.002779Z","shell.execute_reply":"2022-07-12T01:14:49.010584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfTrainModified.text[1191], trainDf.text[1191]","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:49.014535Z","iopub.execute_input":"2022-07-12T01:14:49.014874Z","iopub.status.idle":"2022-07-12T01:14:49.025891Z","shell.execute_reply.started":"2022-07-12T01:14:49.014845Z","shell.execute_reply":"2022-07-12T01:14:49.024348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Change each row of the tweets from list into a single string","metadata":{}},{"cell_type":"code","source":"dfTrainModified['text']=dfTrainModified['text'].apply(lambda x: ' '.join(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:49.027214Z","iopub.execute_input":"2022-07-12T01:14:49.027540Z","iopub.status.idle":"2022-07-12T01:14:49.045005Z","shell.execute_reply.started":"2022-07-12T01:14:49.027512Z","shell.execute_reply":"2022-07-12T01:14:49.044057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfTrainModified.text[1191], trainDf.text[1191]","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:49.048901Z","iopub.execute_input":"2022-07-12T01:14:49.049588Z","iopub.status.idle":"2022-07-12T01:14:49.057055Z","shell.execute_reply.started":"2022-07-12T01:14:49.049542Z","shell.execute_reply":"2022-07-12T01:14:49.055942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Preprocessing Pipline Function","metadata":{}},{"cell_type":"code","source":"def preprocessPipline(data, labelName):\n    \n    dataTemp = data.copy()\n    dataTemp[labelName] = dataTemp[labelName].apply(lambda x: re.sub('(\\s*)@\\w+(\\s*)','', x))\n    dataTemp[labelName] = dataTemp[labelName].apply(lambda x: re.sub('\\d+','', x))\n    dataTemp[labelName] = dataTemp[labelName].apply(lambda x: re.sub('#','', x))\n    dataTemp[labelName] = dataTemp[labelName].apply(lambda x: re.sub('https?://\\S+|www\\.\\S+','',x))\n    dataTemp[labelName] = dataTemp[labelName].apply(lambda x: re.sub('[^A-Za-z]',' ',x))\n    dataTemp[labelName] = dataTemp[labelName].apply(lambda x: x.lower())\n    dataTemp[labelName] = dataTemp[labelName].apply(lambda x : [word for word in x.split()  if word not in stop_words])\n    dataTemp[labelName] = dataTemp[labelName].apply(lambda x: [lemmatizer.lemmatize(word) for word in x])\n    dataTemp[labelName] = dataTemp[labelName].apply(lambda x: ' '.join(x))\n    return dataTemp","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:49.058818Z","iopub.execute_input":"2022-07-12T01:14:49.059674Z","iopub.status.idle":"2022-07-12T01:14:49.068813Z","shell.execute_reply.started":"2022-07-12T01:14:49.059627Z","shell.execute_reply":"2022-07-12T01:14:49.067548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(dfTrainModified.head(2))\ndisplay(preprocessPipline(trainDf, 'text').head(2))\ndisplay(trainDf.head(2))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:49.070086Z","iopub.execute_input":"2022-07-12T01:14:49.070453Z","iopub.status.idle":"2022-07-12T01:14:49.682351Z","shell.execute_reply.started":"2022-07-12T01:14:49.070413Z","shell.execute_reply":"2022-07-12T01:14:49.681310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Bag Of Words","metadata":{}},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\n\nvectorizer=CountVectorizer(max_features=10000,ngram_range=(1,2))\nvectorizer.fit(dfTrainModified['text'].tolist())\nx = vectorizer.transform(dfTrainModified['text'].tolist())\ncolumns = vectorizer.get_feature_names()\nfinalDF = pd.DataFrame(x.todense(), columns=columns, index=dfTrainModified['text'].tolist())\n\nfinalDF.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:49.683432Z","iopub.execute_input":"2022-07-12T01:14:49.684554Z","iopub.status.idle":"2022-07-12T01:14:50.253598Z","shell.execute_reply.started":"2022-07-12T01:14:49.684507Z","shell.execute_reply":"2022-07-12T01:14:50.252442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_data = dfTrainModified['target']\n\nfrom sklearn.model_selection import train_test_split\nx_train, x_val ,y_train ,y_val = train_test_split(finalDF,Y_data,test_size=0.2,stratify=Y_data)\nx_train.shape, x_val.shape, y_train.shape, y_val.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:50.258215Z","iopub.execute_input":"2022-07-12T01:14:50.258858Z","iopub.status.idle":"2022-07-12T01:14:50.649594Z","shell.execute_reply.started":"2022-07-12T01:14:50.258826Z","shell.execute_reply":"2022-07-12T01:14:50.648294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Standard Neural Network","metadata":{}},{"cell_type":"code","source":"from keras import backend as K\n\ndef recall_m(y_true, y_pred):\n    true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n    possible_positives = K.sum(K.round(K.clip(y_true, 0, 1)))\n    recall = true_positives / (possible_positives + K.epsilon())\n    return recall\n\ndef precision_m(y_true, y_pred):\n    true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n    predicted_positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n    precision = true_positives / (predicted_positives + K.epsilon())\n    return precision\n\ndef f1_m(y_true, y_pred):\n    precision = precision_m(y_true, y_pred)\n    recall = recall_m(y_true, y_pred)\n    return 2*((precision*recall)/(precision+recall+K.epsilon()))\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:50.651075Z","iopub.execute_input":"2022-07-12T01:14:50.651535Z","iopub.status.idle":"2022-07-12T01:14:59.932504Z","shell.execute_reply.started":"2022-07-12T01:14:50.651492Z","shell.execute_reply":"2022-07-12T01:14:59.931404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom keras import layers,models\n\nmodel_1=models.Sequential()\nmodel_1.add(layers.Dense(128,activation='relu',input_shape=(10000,)))# 10000 is my input features from BOW\nmodel_1.add(layers.Dense(64,activation='relu'))\nmodel_1.add(layers.Dense(32,activation='relu'))\nmodel_1.add(layers.Dense(1,activation='sigmoid'))\nmodel_1.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:14:59.934057Z","iopub.execute_input":"2022-07-12T01:14:59.934669Z","iopub.status.idle":"2022-07-12T01:15:00.100968Z","shell.execute_reply.started":"2022-07-12T01:14:59.934636Z","shell.execute_reply":"2022-07-12T01:15:00.099794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_trainArr = np.asarray(x_train)\nx_valArr = np.asarray(x_val)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:15:00.102251Z","iopub.execute_input":"2022-07-12T01:15:00.103054Z","iopub.status.idle":"2022-07-12T01:15:00.109338Z","shell.execute_reply.started":"2022-07-12T01:15:00.103019Z","shell.execute_reply":"2022-07-12T01:15:00.108272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import keras \nfrom tensorflow.keras.optimizers import Adam\n\nmodel_1.compile(optimizer= 'adam',\n              loss= 'binary_crossentropy',\n              metrics=['acc', f1_m])\n\nhistory = model_1.fit(x_trainArr,y_train,epochs=10,validation_data=(x_valArr,y_val))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:15:00.111503Z","iopub.execute_input":"2022-07-12T01:15:00.112322Z","iopub.status.idle":"2022-07-12T01:15:22.638085Z","shell.execute_reply.started":"2022-07-12T01:15:00.112276Z","shell.execute_reply":"2022-07-12T01:15:22.636692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## GRU","metadata":{}},{"cell_type":"markdown","source":"- Check various lengths of tweets","metadata":{}},{"cell_type":"code","source":"tweetLength = {}\n\nfor i in dfTrainModified.text.apply(lambda x:x.split()):\n    \n    if '{}'.format(len(i)) in tweetLength.keys():\n        \n        tweetLength['{}'.format(len(i))] += 1\n    else:\n        tweetLength['{}'.format(len(i))] = 1\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:15:22.640166Z","iopub.execute_input":"2022-07-12T01:15:22.640936Z","iopub.status.idle":"2022-07-12T01:15:22.671863Z","shell.execute_reply.started":"2022-07-12T01:15:22.640894Z","shell.execute_reply":"2022-07-12T01:15:22.669751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tweetLengthDf = pd.DataFrame.from_dict(tweetLength, orient='index').reset_index()\ntweetLengthDf = tweetLengthDf.rename(columns={'index': 'tweetLen', 0: 'freq'})\ntweetLengthDf['tweetLen'] = pd.to_numeric(tweetLengthDf['tweetLen'])\n\ntweetLengthDf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:15:22.674676Z","iopub.execute_input":"2022-07-12T01:15:22.675622Z","iopub.status.idle":"2022-07-12T01:15:22.697737Z","shell.execute_reply.started":"2022-07-12T01:15:22.675573Z","shell.execute_reply":"2022-07-12T01:15:22.696480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px\nfig = px.bar(tweetLengthDf.sort_values('tweetLen'), x='tweetLen', y='freq')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:15:22.699997Z","iopub.execute_input":"2022-07-12T01:15:22.701252Z","iopub.status.idle":"2022-07-12T01:15:24.694603Z","shell.execute_reply.started":"2022-07-12T01:15:22.701183Z","shell.execute_reply":"2022-07-12T01:15:24.693268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# since the longest tweet length is not very long(23) I wont make padding and lets see what will happen\nmaxTweetLen = 23\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:15:24.696049Z","iopub.execute_input":"2022-07-12T01:15:24.696485Z","iopub.status.idle":"2022-07-12T01:15:24.700811Z","shell.execute_reply.started":"2022-07-12T01:15:24.696451Z","shell.execute_reply":"2022-07-12T01:15:24.700074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"processedDf = preprocessPipline(trainDf, 'text')\nprocessedDf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:15:24.701770Z","iopub.execute_input":"2022-07-12T01:15:24.702506Z","iopub.status.idle":"2022-07-12T01:15:25.317161Z","shell.execute_reply.started":"2022-07-12T01:15:24.702473Z","shell.execute_reply":"2022-07-12T01:15:25.315966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.text import Tokenizer\n\ntokenizer = Tokenizer(oov_token=\"unk\")\ntokenizer.fit_on_texts(processedDf.text)\ntokenizer_seq = tokenizer.texts_to_sequences(processedDf.text)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:15:25.318778Z","iopub.execute_input":"2022-07-12T01:15:25.319252Z","iopub.status.idle":"2022-07-12T01:15:25.532490Z","shell.execute_reply.started":"2022-07-12T01:15:25.319205Z","shell.execute_reply":"2022-07-12T01:15:25.531288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.sequence import pad_sequences\n\ntokenizer_seq = pad_sequences(tokenizer_seq, maxlen=23, padding='post',truncating='post')\ntokenizer_seq","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:15:25.534064Z","iopub.execute_input":"2022-07-12T01:15:25.534890Z","iopub.status.idle":"2022-07-12T01:15:25.569241Z","shell.execute_reply.started":"2022-07-12T01:15:25.534840Z","shell.execute_reply":"2022-07-12T01:15:25.568353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer_seq[2]","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:15:25.570766Z","iopub.execute_input":"2022-07-12T01:15:25.571503Z","iopub.status.idle":"2022-07-12T01:15:25.579601Z","shell.execute_reply.started":"2022-07-12T01:15:25.571456Z","shell.execute_reply":"2022-07-12T01:15:25.578219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab_size = len(tokenizer.word_index)\nvocab_size","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:15:25.580868Z","iopub.execute_input":"2022-07-12T01:15:25.581936Z","iopub.status.idle":"2022-07-12T01:15:25.593880Z","shell.execute_reply.started":"2022-07-12T01:15:25.581888Z","shell.execute_reply":"2022-07-12T01:15:25.592694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_data = processedDf['target']\n\nfrom sklearn.model_selection import train_test_split\nx_train, x_val ,y_train ,y_val = train_test_split(tokenizer_seq, Y_data, test_size=0.2, stratify=Y_data)\nx_train.shape, x_val.shape, y_train.shape, y_val.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:15:25.595610Z","iopub.execute_input":"2022-07-12T01:15:25.596229Z","iopub.status.idle":"2022-07-12T01:15:25.612057Z","shell.execute_reply.started":"2022-07-12T01:15:25.596152Z","shell.execute_reply":"2022-07-12T01:15:25.610688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_GRU = models.Sequential()\nmodel_GRU.add(layers.Embedding(vocab_size + 1, 64))\nmodel_GRU.add(layers.Dropout(0.2))\nmodel_GRU.add(layers.GRU(64))\nmodel_GRU.add(layers.Dense(1, activation='sigmoid'))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:15:25.613528Z","iopub.execute_input":"2022-07-12T01:15:25.614369Z","iopub.status.idle":"2022-07-12T01:15:25.877510Z","shell.execute_reply.started":"2022-07-12T01:15:25.614333Z","shell.execute_reply":"2022-07-12T01:15:25.876301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_GRU.compile(optimizer= 'adam',\n              loss= 'binary_crossentropy',\n              metrics=['acc', f1_m])\nmodel_GRU.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:15:25.879021Z","iopub.execute_input":"2022-07-12T01:15:25.879396Z","iopub.status.idle":"2022-07-12T01:15:25.893533Z","shell.execute_reply.started":"2022-07-12T01:15:25.879362Z","shell.execute_reply":"2022-07-12T01:15:25.892051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model_GRU.fit(x_train, y_train, epochs=10, validation_data=(x_val, y_val))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:15:25.895141Z","iopub.execute_input":"2022-07-12T01:15:25.896139Z","iopub.status.idle":"2022-07-12T01:16:32.992477Z","shell.execute_reply.started":"2022-07-12T01:15:25.896100Z","shell.execute_reply":"2022-07-12T01:16:32.991533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## GRU Bidirectional ","metadata":{}},{"cell_type":"code","source":"model_gruBid = models.Sequential()\nmodel_gruBid.add(layers.Embedding(vocab_size +1 ,64))\nmodel_gruBid.add(layers.Dropout(0.2))\nmodel_gruBid.add(layers.Bidirectional(layers.GRU(64,return_sequences=True)))\nmodel_gruBid.add(layers.Dropout(0.2))\nmodel_gruBid.add(layers.Bidirectional(layers.GRU(64)))\nmodel_gruBid.add(layers.Dense(1, activation='sigmoid'))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:16:32.997441Z","iopub.execute_input":"2022-07-12T01:16:32.998043Z","iopub.status.idle":"2022-07-12T01:16:33.802557Z","shell.execute_reply.started":"2022-07-12T01:16:32.998004Z","shell.execute_reply":"2022-07-12T01:16:33.801322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_gruBid.compile(optimizer= 'adam',\n              loss= 'binary_crossentropy',\n              metrics=['acc', f1_m])\nmodel_gruBid.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:16:33.803986Z","iopub.execute_input":"2022-07-12T01:16:33.804873Z","iopub.status.idle":"2022-07-12T01:16:33.817888Z","shell.execute_reply.started":"2022-07-12T01:16:33.804820Z","shell.execute_reply":"2022-07-12T01:16:33.816468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model_gruBid.fit(x_train, y_train, epochs=10, validation_data=(x_val, y_val))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:16:33.819183Z","iopub.execute_input":"2022-07-12T01:16:33.819627Z","iopub.status.idle":"2022-07-12T01:19:47.963869Z","shell.execute_reply.started":"2022-07-12T01:16:33.819594Z","shell.execute_reply":"2022-07-12T01:19:47.962658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## LSTM","metadata":{}},{"cell_type":"code","source":"model_lstm = models.Sequential()\nmodel_lstm.add(layers.Embedding(vocab_size + 1,64))\nmodel_lstm.add(layers.Dropout(0.5))\nmodel_lstm.add(layers.LSTM(64))\nmodel_lstm.add(layers.Dense(1, activation='sigmoid'))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:19:47.965889Z","iopub.execute_input":"2022-07-12T01:19:47.966728Z","iopub.status.idle":"2022-07-12T01:19:48.209362Z","shell.execute_reply.started":"2022-07-12T01:19:47.966680Z","shell.execute_reply":"2022-07-12T01:19:48.208269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.optimizers import RMSprop\n\nmodel_lstm.compile(optimizer= RMSprop(lr=0.0001),\n              loss= 'binary_crossentropy',\n              metrics=['acc', f1_m])\nmodel_lstm.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:19:48.211008Z","iopub.execute_input":"2022-07-12T01:19:48.211437Z","iopub.status.idle":"2022-07-12T01:19:48.226361Z","shell.execute_reply.started":"2022-07-12T01:19:48.211395Z","shell.execute_reply":"2022-07-12T01:19:48.225139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model_lstm.fit(x_train, y_train, epochs=10, validation_data=(x_val, y_val))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:19:48.228182Z","iopub.execute_input":"2022-07-12T01:19:48.228653Z","iopub.status.idle":"2022-07-12T01:21:12.065728Z","shell.execute_reply.started":"2022-07-12T01:19:48.228611Z","shell.execute_reply":"2022-07-12T01:21:12.064435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## LSTM Bidirectional","metadata":{}},{"cell_type":"code","source":"model_lstmBid = models.Sequential()\nmodel_lstmBid.add(layers.Embedding(vocab_size + 1,128))\nmodel_lstmBid.add(layers.Dropout(0.4))\nmodel_lstmBid.add(layers.Bidirectional(layers.LSTM(64)))\nmodel_lstmBid.add(layers.Dense(1, activation='sigmoid'))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:21:12.068554Z","iopub.execute_input":"2022-07-12T01:21:12.069076Z","iopub.status.idle":"2022-07-12T01:21:12.536955Z","shell.execute_reply.started":"2022-07-12T01:21:12.069027Z","shell.execute_reply":"2022-07-12T01:21:12.535827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.optimizers import RMSprop\n\nmodel_lstmBid.compile(optimizer= RMSprop(lr=0.0001),\n              loss= 'binary_crossentropy',\n              metrics=['acc', f1_m])\nmodel_lstmBid.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:21:12.538457Z","iopub.execute_input":"2022-07-12T01:21:12.538790Z","iopub.status.idle":"2022-07-12T01:21:12.551295Z","shell.execute_reply.started":"2022-07-12T01:21:12.538760Z","shell.execute_reply":"2022-07-12T01:21:12.550206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model_lstmBid.fit(x_train, y_train, epochs=10, validation_data=(x_val, y_val))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:21:12.552430Z","iopub.execute_input":"2022-07-12T01:21:12.553386Z","iopub.status.idle":"2022-07-12T01:22:45.742008Z","shell.execute_reply.started":"2022-07-12T01:21:12.553353Z","shell.execute_reply":"2022-07-12T01:22:45.740786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt \nacc = history.history['acc']\nval_acc = history.history['val_acc']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nepochs = range(1, len(acc) + 1)\n\n# \"bo\" is for \"blue dot\"\nplt.plot(epochs, loss, 'bo', label='Training loss')\n# b is for \"solid blue line\"\nplt.plot(epochs, val_loss, 'b', label='Validation loss')\nplt.title('Training and validation loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:22:45.743616Z","iopub.execute_input":"2022-07-12T01:22:45.743933Z","iopub.status.idle":"2022-07-12T01:22:45.980516Z","shell.execute_reply.started":"2022-07-12T01:22:45.743897Z","shell.execute_reply":"2022-07-12T01:22:45.979665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc = history.history['acc']\nval_acc = history.history['val_acc']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nplt.plot(epochs, acc, 'bo', label='Training acc')\nplt.plot(epochs, val_acc, 'b', label='Validation acc')\nplt.title('Training and validation accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('accuracy')\nplt.legend()\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:22:45.981654Z","iopub.execute_input":"2022-07-12T01:22:45.982121Z","iopub.status.idle":"2022-07-12T01:22:46.179392Z","shell.execute_reply.started":"2022-07-12T01:22:45.982087Z","shell.execute_reply":"2022-07-12T01:22:46.178550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1 = history.history['f1_m']\nvalF1 = history.history['val_f1_m']\n\nplt.plot(epochs, f1, 'bo', label='Training f1')\nplt.plot(epochs, valF1, 'b', label='Validation f1')\nplt.title('Training and validation f1')\nplt.xlabel('Epochs')\nplt.ylabel('f1')\nplt.legend()\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:22:46.180535Z","iopub.execute_input":"2022-07-12T01:22:46.180997Z","iopub.status.idle":"2022-07-12T01:22:46.334817Z","shell.execute_reply.started":"2022-07-12T01:22:46.180968Z","shell.execute_reply":"2022-07-12T01:22:46.333983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"finalTestDf = preprocessPipline(testDf, 'text')\n\ndisplay(finalTestDf.head(2))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T02:17:39.709609Z","iopub.execute_input":"2022-07-12T02:17:39.710425Z","iopub.status.idle":"2022-07-12T02:17:40.011398Z","shell.execute_reply.started":"2022-07-12T02:17:39.710384Z","shell.execute_reply":"2022-07-12T02:17:40.010262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"finalTestDf = tokenizer.texts_to_sequences(finalTestDf.text)\n\nfinalTestDfPadded = pad_sequences(finalTestDf,\n                                maxlen=23, \n                                truncating='post', \n                                padding='post'\n                               )","metadata":{"execution":{"iopub.status.busy":"2022-07-12T02:17:40.112104Z","iopub.execute_input":"2022-07-12T02:17:40.112520Z","iopub.status.idle":"2022-07-12T02:17:40.169120Z","shell.execute_reply.started":"2022-07-12T02:17:40.112486Z","shell.execute_reply":"2022-07-12T02:17:40.167726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result = model_lstmBid.predict(finalTestDfPadded)\nresult[result>=0.5]=1\nresult[result<0.5]=0\nresult","metadata":{"execution":{"iopub.status.busy":"2022-07-12T02:17:50.246148Z","iopub.execute_input":"2022-07-12T02:17:50.248693Z","iopub.status.idle":"2022-07-12T02:17:51.461867Z","shell.execute_reply.started":"2022-07-12T02:17:50.248603Z","shell.execute_reply":"2022-07-12T02:17:51.461034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission=pd.read_csv('../input/nlp-getting-started/sample_submission.csv')\nsample_submission['target']=result\nsample_submission['target']=sample_submission['target'].astype(int)\n#sample_submission.head()\nsample_submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T02:17:57.502600Z","iopub.execute_input":"2022-07-12T02:17:57.503491Z","iopub.status.idle":"2022-07-12T02:17:57.526218Z","shell.execute_reply.started":"2022-07-12T02:17:57.503438Z","shell.execute_reply":"2022-07-12T02:17:57.524972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}