{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-07T12:13:35.856276Z","iopub.execute_input":"2022-07-07T12:13:35.856754Z","iopub.status.idle":"2022-07-07T12:13:35.867085Z","shell.execute_reply.started":"2022-07-07T12:13:35.856723Z","shell.execute_reply":"2022-07-07T12:13:35.865590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nfrom collections import defaultdict\nfrom nltk.corpus import stopwords\nstop=set(stopwords.words('english'))\n\nimport plotly.graph_objs as go\nimport plotly.offline as py\nimport re\nimport string\nimport nltk\nfrom nltk.util import ngrams\n\n\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:13:37.276085Z","iopub.execute_input":"2022-07-07T12:13:37.277246Z","iopub.status.idle":"2022-07-07T12:13:37.286777Z","shell.execute_reply.started":"2022-07-07T12:13:37.277180Z","shell.execute_reply":"2022-07-07T12:13:37.285405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tweet= pd.read_csv('../input/nlp-getting-started/train.csv')\ntest=pd.read_csv('../input/nlp-getting-started/test.csv')\ntweet.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:13:40.092000Z","iopub.execute_input":"2022-07-07T12:13:40.093119Z","iopub.status.idle":"2022-07-07T12:13:40.139973Z","shell.execute_reply.started":"2022-07-07T12:13:40.093085Z","shell.execute_reply":"2022-07-07T12:13:40.138717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tweet.count()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:13:41.691136Z","iopub.execute_input":"2022-07-07T12:13:41.691862Z","iopub.status.idle":"2022-07-07T12:13:41.707090Z","shell.execute_reply.started":"2022-07-07T12:13:41.691829Z","shell.execute_reply":"2022-07-07T12:13:41.705636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tweet.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:13:44.103219Z","iopub.execute_input":"2022-07-07T12:13:44.103687Z","iopub.status.idle":"2022-07-07T12:13:44.122795Z","shell.execute_reply.started":"2022-07-07T12:13:44.103656Z","shell.execute_reply":"2022-07-07T12:13:44.121297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tweet.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:13:46.227483Z","iopub.execute_input":"2022-07-07T12:13:46.227922Z","iopub.status.idle":"2022-07-07T12:13:46.236865Z","shell.execute_reply.started":"2022-07-07T12:13:46.227892Z","shell.execute_reply":"2022-07-07T12:13:46.235396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tweet.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:13:48.108456Z","iopub.execute_input":"2022-07-07T12:13:48.108830Z","iopub.status.idle":"2022-07-07T12:13:48.125969Z","shell.execute_reply.started":"2022-07-07T12:13:48.108794Z","shell.execute_reply":"2022-07-07T12:13:48.124687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tweet.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:13:49.751749Z","iopub.execute_input":"2022-07-07T12:13:49.752112Z","iopub.status.idle":"2022-07-07T12:13:49.760553Z","shell.execute_reply.started":"2022-07-07T12:13:49.752081Z","shell.execute_reply":"2022-07-07T12:13:49.759143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_f = [f for f in tweet.columns if tweet[f].dtype == 'object' ]","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:13:51.301253Z","iopub.execute_input":"2022-07-07T12:13:51.302336Z","iopub.status.idle":"2022-07-07T12:13:51.308976Z","shell.execute_reply.started":"2022-07-07T12:13:51.302289Z","shell.execute_reply":"2022-07-07T12:13:51.307569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_f","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:13:53.231695Z","iopub.execute_input":"2022-07-07T12:13:53.232565Z","iopub.status.idle":"2022-07-07T12:13:53.242217Z","shell.execute_reply.started":"2022-07-07T12:13:53.232517Z","shell.execute_reply":"2022-07-07T12:13:53.240760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tweet['word_count'] = tweet['text'].apply(lambda x: len(str(x).split()))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:13:54.712082Z","iopub.execute_input":"2022-07-07T12:13:54.713078Z","iopub.status.idle":"2022-07-07T12:13:54.735680Z","shell.execute_reply.started":"2022-07-07T12:13:54.713026Z","shell.execute_reply":"2022-07-07T12:13:54.734344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,(ax1,ax2)=plt.subplots(1,2,figsize=(10,5))\ntweet_len=tweet[tweet['target']==1]['text'].str.len()\nax1.hist(tweet_len,color='red')\nax1.set_title('disaster tweets')\ntweet_len=tweet[tweet['target']==0]['text'].str.len()\nax2.hist(tweet_len,color='green')\nax2.set_title('Not disaster tweets')\nfig.suptitle('Characters in tweets')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:13:56.385479Z","iopub.execute_input":"2022-07-07T12:13:56.386269Z","iopub.status.idle":"2022-07-07T12:13:56.760835Z","shell.execute_reply.started":"2022-07-07T12:13:56.386202Z","shell.execute_reply":"2022-07-07T12:13:56.759407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,(ax1,ax2)=plt.subplots(1,2,figsize=(10,5))\nword=tweet[tweet['target']==1]['text'].str.split().apply(lambda x : [len(i) for i in x])\nsns.distplot(word.map(lambda x: np.mean(x)),ax=ax1,color='red')\nax1.set_title('disaster')\nword=tweet[tweet['target']==0]['text'].str.split().apply(lambda x : [len(i) for i in x])\nsns.distplot(word.map(lambda x: np.mean(x)),ax=ax2,color='green')\nax2.set_title('Not disaster')\nfig.suptitle('Average word length in each tweet')","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:13:58.172003Z","iopub.execute_input":"2022-07-07T12:13:58.174011Z","iopub.status.idle":"2022-07-07T12:13:59.000583Z","shell.execute_reply.started":"2022-07-07T12:13:58.173962Z","shell.execute_reply":"2022-07-07T12:13:58.999173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_corpus(target):\n    corpus=[]\n    \n    for x in tweet[tweet['target']==target]['text'].str.split():\n        for i in x:\n            corpus.append(i)\n    return corpus","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:14:02.023601Z","iopub.execute_input":"2022-07-07T12:14:02.024498Z","iopub.status.idle":"2022-07-07T12:14:02.031222Z","shell.execute_reply.started":"2022-07-07T12:14:02.024462Z","shell.execute_reply":"2022-07-07T12:14:02.029571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corpus=create_corpus(0)\n\ndic=defaultdict(int)\nfor word in corpus:\n    if word in stop:\n        dic[word]+=1\n        \ntop=sorted(dic.items(), key=lambda x:x[1],reverse=True)[:10] ","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:14:03.896104Z","iopub.execute_input":"2022-07-07T12:14:03.896843Z","iopub.status.idle":"2022-07-07T12:14:03.937716Z","shell.execute_reply.started":"2022-07-07T12:14:03.896807Z","shell.execute_reply":"2022-07-07T12:14:03.936541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x,y=zip(*top)\nplt.bar(x,y)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:14:07.731218Z","iopub.execute_input":"2022-07-07T12:14:07.731626Z","iopub.status.idle":"2022-07-07T12:14:07.971647Z","shell.execute_reply.started":"2022-07-07T12:14:07.731595Z","shell.execute_reply":"2022-07-07T12:14:07.970483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corpus=create_corpus(1)\n\ndic=defaultdict(int)\nfor word in corpus:\n    if word in stop:\n        dic[word]+=1\n\ntop=sorted(dic.items(), key=lambda x:x[1],reverse=True)[:10] \n    \n\n\nx,y=zip(*top)\nplt.bar(x,y)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:14:09.371794Z","iopub.execute_input":"2022-07-07T12:14:09.372648Z","iopub.status.idle":"2022-07-07T12:14:09.613515Z","shell.execute_reply.started":"2022-07-07T12:14:09.372585Z","shell.execute_reply":"2022-07-07T12:14:09.612215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cnt_ = tweet['location'].value_counts()\ncnt_.reset_index()\ncnt_ = cnt_[:20,]\ntrace1 = go.Bar(\n                x = cnt_.index,\n                y = cnt_.values,\n                name = \"Number of tweets in dataset according to location\",\n                marker = dict(color = 'rgba(255, 0, 100, 0.5)',\n                             line=dict(color='rgb(0,0,0)',width=1.5)),\n                )\n\ndata = [trace1]\nlayout = go.Layout(barmode = \"group\",title = 'Number of tweets in dataset according to location')\nfig = go.Figure(data = data, layout = layout)\npy.iplot(fig)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:14:10.817219Z","iopub.execute_input":"2022-07-07T12:14:10.817979Z","iopub.status.idle":"2022-07-07T12:14:10.863870Z","shell.execute_reply.started":"2022-07-07T12:14:10.817943Z","shell.execute_reply":"2022-07-07T12:14:10.862411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=pd.concat([tweet,test])\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:14:12.347658Z","iopub.execute_input":"2022-07-07T12:14:12.348502Z","iopub.status.idle":"2022-07-07T12:14:12.360747Z","shell.execute_reply.started":"2022-07-07T12:14:12.348466Z","shell.execute_reply":"2022-07-07T12:14:12.359256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:14:13.899672Z","iopub.execute_input":"2022-07-07T12:14:13.900046Z","iopub.status.idle":"2022-07-07T12:14:13.924942Z","shell.execute_reply.started":"2022-07-07T12:14:13.900013Z","shell.execute_reply":"2022-07-07T12:14:13.923632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_URL(text):\n    url = re.compile(r'https?://\\S+|www\\.\\S+')\n    return url.sub(r'',text)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:14:18.239911Z","iopub.execute_input":"2022-07-07T12:14:18.241113Z","iopub.status.idle":"2022-07-07T12:14:18.247804Z","shell.execute_reply.started":"2022-07-07T12:14:18.241080Z","shell.execute_reply":"2022-07-07T12:14:18.246497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tweet['text']=tweet['text'].apply(lambda x : remove_URL(x))\ntest['text']=test['text'].apply(lambda x : remove_URL(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:14:21.427414Z","iopub.execute_input":"2022-07-07T12:14:21.427808Z","iopub.status.idle":"2022-07-07T12:14:21.480809Z","shell.execute_reply.started":"2022-07-07T12:14:21.427777Z","shell.execute_reply":"2022-07-07T12:14:21.479572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example = \"\"\"<div>\n<h1>Real or Fake</h1>\n<p>Kaggle </p>\n<a href=\"https://www.kaggle.com/c/nlp-getting-started\">getting started</a>\n</div>\"\"\"","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:14:22.951800Z","iopub.execute_input":"2022-07-07T12:14:22.952312Z","iopub.status.idle":"2022-07-07T12:14:22.958652Z","shell.execute_reply.started":"2022-07-07T12:14:22.952235Z","shell.execute_reply":"2022-07-07T12:14:22.957191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_html(text):\n    html=re.compile(r'<.*?>')\n    return html.sub(r'',text)\nprint(remove_html(example))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:14:24.191955Z","iopub.execute_input":"2022-07-07T12:14:24.192369Z","iopub.status.idle":"2022-07-07T12:14:24.199809Z","shell.execute_reply.started":"2022-07-07T12:14:24.192307Z","shell.execute_reply":"2022-07-07T12:14:24.198149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tweet['text']=tweet['text'].apply(lambda x : remove_html(x))\ntest['text']=test['text'].apply(lambda x : remove_html(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:14:25.451624Z","iopub.execute_input":"2022-07-07T12:14:25.451977Z","iopub.status.idle":"2022-07-07T12:14:25.480811Z","shell.execute_reply.started":"2022-07-07T12:14:25.451947Z","shell.execute_reply":"2022-07-07T12:14:25.479589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_emoji(text):\n    emoji_pattern = re.compile(\"[\"\n                           u\"\\U0001F600-\\U0001F64F\"  # emoticons\n                           u\"\\U0001F300-\\U0001F5FF\"  # symbols & pictographs\n                           u\"\\U0001F680-\\U0001F6FF\"  # transport & map symbols\n                           u\"\\U0001F1E0-\\U0001F1FF\"  # flags (iOS)\n                           u\"\\U00002702-\\U000027B0\"\n                           u\"\\U000024C2-\\U0001F251\"\n                           \"]+\", flags=re.UNICODE)\n    return emoji_pattern.sub(r'', text)\n\nremove_emoji(\"Omg another Earthquake 😔😔\")","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:14:26.831542Z","iopub.execute_input":"2022-07-07T12:14:26.832697Z","iopub.status.idle":"2022-07-07T12:14:26.843610Z","shell.execute_reply.started":"2022-07-07T12:14:26.832651Z","shell.execute_reply":"2022-07-07T12:14:26.842152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tweet['text']=tweet['text'].apply(lambda x: remove_emoji(x))\ntest['text']=test['text'].apply(lambda x: remove_emoji(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:14:28.896147Z","iopub.execute_input":"2022-07-07T12:14:28.896732Z","iopub.status.idle":"2022-07-07T12:14:28.975273Z","shell.execute_reply.started":"2022-07-07T12:14:28.896702Z","shell.execute_reply":"2022-07-07T12:14:28.973987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_punct(text):\n    table=str.maketrans('','',string.punctuation)\n    return text.translate(table)\n\nexample=\"I am a #king\"\nprint(remove_punct(example))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:14:30.536299Z","iopub.execute_input":"2022-07-07T12:14:30.536706Z","iopub.status.idle":"2022-07-07T12:14:30.545063Z","shell.execute_reply.started":"2022-07-07T12:14:30.536678Z","shell.execute_reply":"2022-07-07T12:14:30.543513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tweet['text']=tweet['text'].apply(lambda x : remove_punct(x))\ntest['text']=test['text'].apply(lambda x : remove_punct(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:14:31.848986Z","iopub.execute_input":"2022-07-07T12:14:31.849529Z","iopub.status.idle":"2022-07-07T12:14:31.918567Z","shell.execute_reply.started":"2022-07-07T12:14:31.849496Z","shell.execute_reply":"2022-07-07T12:14:31.917190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.sample(50)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:14:33.276676Z","iopub.execute_input":"2022-07-07T12:14:33.277082Z","iopub.status.idle":"2022-07-07T12:14:33.300797Z","shell.execute_reply.started":"2022-07-07T12:14:33.277033Z","shell.execute_reply":"2022-07-07T12:14:33.299461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install pyspellchecker","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:14:34.572672Z","iopub.execute_input":"2022-07-07T12:14:34.574630Z","iopub.status.idle":"2022-07-07T12:14:45.977051Z","shell.execute_reply.started":"2022-07-07T12:14:34.574579Z","shell.execute_reply":"2022-07-07T12:14:45.975287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from spellchecker import SpellChecker\n\nspell = SpellChecker()\ndef correct_spellings(text):\n    corrected_text = []\n    misspelled_words = spell.unknown(text.split())\n    for word in text.split():\n        if word in misspelled_words:\n            corrected_text.append(spell.correction(word))\n        else:\n            corrected_text.append(word)\n    return \" \".join(corrected_text)\n        \ntext = \"brek corect plese\"\ncorrect_spellings(text)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:14:45.979408Z","iopub.execute_input":"2022-07-07T12:14:45.980286Z","iopub.status.idle":"2022-07-07T12:14:46.136806Z","shell.execute_reply.started":"2022-07-07T12:14:45.980205Z","shell.execute_reply":"2022-07-07T12:14:46.135283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"real = tweet[tweet.target==1].reset_index()\nfake = tweet[tweet.target==0].reset_index()\ntype(real)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:14:46.139460Z","iopub.execute_input":"2022-07-07T12:14:46.140750Z","iopub.status.idle":"2022-07-07T12:14:46.158916Z","shell.execute_reply.started":"2022-07-07T12:14:46.140704Z","shell.execute_reply":"2022-07-07T12:14:46.157369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_ngrams(data,n):\n    all_words = []\n    for i in range(len(data)):\n        temp = data[\"text\"][i].split()\n        for word in temp:\n            all_words.append(word)\n\n    tokenized = all_words\n    esBigrams = ngrams(tokenized, n)\n\n    esBigram_wordlist = nltk.FreqDist(esBigrams)\n    top100 = esBigram_wordlist.most_common(100)\n    top100 = dict(top100)\n    df_ngrams = pd.DataFrame(sorted(top100.items(), key=lambda x: x[1])[::-1])\n    return df_ngrams","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:14:47.811794Z","iopub.execute_input":"2022-07-07T12:14:47.812789Z","iopub.status.idle":"2022-07-07T12:14:47.822353Z","shell.execute_reply.started":"2022-07-07T12:14:47.812739Z","shell.execute_reply":"2022-07-07T12:14:47.820879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def draw_barplots(real,fake,title):\n#     plt.figure(figsize = (40,80),dpi=100)\n\n    plt.subplot(1,2,1)\n    sns.barplot(y=real[0].values[:4], x=real[1].values[:4], color='green')\n    plt.title(\"Top 4\" + title + \"in Real Tweets\",fontsize=15)\n    \n    plt.subplot(1,2,2)\n    sns.barplot(y=fake[0].values[:4], x=fake[1].values[:4],color='red')\n    plt.title(\"Top 4\" + title + \"in Fake Tweets\",fontsize=15)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:14:49.799585Z","iopub.execute_input":"2022-07-07T12:14:49.799952Z","iopub.status.idle":"2022-07-07T12:14:49.811681Z","shell.execute_reply.started":"2022-07-07T12:14:49.799922Z","shell.execute_reply":"2022-07-07T12:14:49.810472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"real_unigrams = get_ngrams(real,1)\nfake_unigrams = get_ngrams(fake,1)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:14:51.675967Z","iopub.execute_input":"2022-07-07T12:14:51.676404Z","iopub.status.idle":"2022-07-07T12:14:51.912888Z","shell.execute_reply.started":"2022-07-07T12:14:51.676374Z","shell.execute_reply":"2022-07-07T12:14:51.911583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"draw_barplots(real_unigrams,fake_unigrams,\" Unigrams \")","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:14:53.064477Z","iopub.execute_input":"2022-07-07T12:14:53.065314Z","iopub.status.idle":"2022-07-07T12:14:53.441427Z","shell.execute_reply.started":"2022-07-07T12:14:53.065251Z","shell.execute_reply":"2022-07-07T12:14:53.440191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"real_bigrams = get_ngrams(real,2)\nfake_bigrams = get_ngrams(fake,2)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:15:02.977125Z","iopub.execute_input":"2022-07-07T12:15:02.977796Z","iopub.status.idle":"2022-07-07T12:15:03.231772Z","shell.execute_reply.started":"2022-07-07T12:15:02.977759Z","shell.execute_reply":"2022-07-07T12:15:03.230526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"draw_barplots(real_bigrams,fake_bigrams,\" Bigrams \")","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:15:04.676176Z","iopub.execute_input":"2022-07-07T12:15:04.676634Z","iopub.status.idle":"2022-07-07T12:15:04.958778Z","shell.execute_reply.started":"2022-07-07T12:15:04.676604Z","shell.execute_reply":"2022-07-07T12:15:04.957296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"real_trigrams = get_ngrams(real,3)\nfake_trigrams = get_ngrams(fake,3)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:15:06.395519Z","iopub.execute_input":"2022-07-07T12:15:06.396717Z","iopub.status.idle":"2022-07-07T12:15:06.652043Z","shell.execute_reply.started":"2022-07-07T12:15:06.396665Z","shell.execute_reply":"2022-07-07T12:15:06.650625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"draw_barplots(real_trigrams,fake_trigrams,\" Trigrams \")","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:15:11.106301Z","iopub.execute_input":"2022-07-07T12:15:11.108337Z","iopub.status.idle":"2022-07-07T12:15:11.597427Z","shell.execute_reply.started":"2022-07-07T12:15:11.108287Z","shell.execute_reply":"2022-07-07T12:15:11.596080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"real[\"text\"][1]\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:15:13.007693Z","iopub.execute_input":"2022-07-07T12:15:13.008385Z","iopub.status.idle":"2022-07-07T12:15:13.019258Z","shell.execute_reply.started":"2022-07-07T12:15:13.008351Z","shell.execute_reply":"2022-07-07T12:15:13.016687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Creating a Model","metadata":{}},{"cell_type":"code","source":"df_train = tweet.drop(['id','keyword','location'],axis=1)\ndf_test = test.drop(['keyword','location'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:15:18.139587Z","iopub.execute_input":"2022-07-07T12:15:18.140718Z","iopub.status.idle":"2022-07-07T12:15:18.150851Z","shell.execute_reply.started":"2022-07-07T12:15:18.140667Z","shell.execute_reply":"2022-07-07T12:15:18.149443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_message = []\ntesting_message = []\ntraining_labels = []\nfor i in range(len(df_train)):\n    training_message.append(df_train.loc[i,'text'])\n    training_labels.append(df_train.loc[i,'target'])\n    \nfor i in range(len(df_test)):\n    testing_message.append(df_test.loc[i,'text'])","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:15:19.860279Z","iopub.execute_input":"2022-07-07T12:15:19.860992Z","iopub.status.idle":"2022-07-07T12:15:20.118747Z","shell.execute_reply.started":"2022-07-07T12:15:19.860959Z","shell.execute_reply":"2022-07-07T12:15:20.117473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab_size = 23000\nembedding_dim = 32\nmax_length = 100\ntrunc_type='post'\npadding_type='post'\noov_tok = \"<OOV>\"","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:15:23.412175Z","iopub.execute_input":"2022-07-07T12:15:23.412636Z","iopub.status.idle":"2022-07-07T12:15:23.419955Z","shell.execute_reply.started":"2022-07-07T12:15:23.412603Z","shell.execute_reply":"2022-07-07T12:15:23.418376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tokenization Process:\ntokenizer = Tokenizer(num_words = vocab_size, oov_token=oov_tok)\ntokenizer.fit_on_texts(training_message)\nword_index = tokenizer.word_index","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:15:27.611900Z","iopub.execute_input":"2022-07-07T12:15:27.612433Z","iopub.status.idle":"2022-07-07T12:15:27.800374Z","shell.execute_reply.started":"2022-07-07T12:15:27.612402Z","shell.execute_reply":"2022-07-07T12:15:27.799133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# texts_to_sequences Process:\ntraining_message = tokenizer.texts_to_sequences(training_message)\ntesting_message = tokenizer.texts_to_sequences(testing_message)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:15:29.435526Z","iopub.execute_input":"2022-07-07T12:15:29.435885Z","iopub.status.idle":"2022-07-07T12:15:29.624322Z","shell.execute_reply.started":"2022-07-07T12:15:29.435855Z","shell.execute_reply":"2022-07-07T12:15:29.623106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#pad_sequences Process:\ntraining_padded = pad_sequences(training_message,maxlen=max_length, truncating=trunc_type, padding=padding_type)\ntesting_padded = pad_sequences(testing_message,maxlen=max_length, truncating=trunc_type, padding=padding_type)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:15:30.863284Z","iopub.execute_input":"2022-07-07T12:15:30.863721Z","iopub.status.idle":"2022-07-07T12:15:30.925152Z","shell.execute_reply.started":"2022-07-07T12:15:30.863692Z","shell.execute_reply":"2022-07-07T12:15:30.923928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_padded = np.array(training_padded)\ntesting_padded = np.array(testing_padded)\ntraining_labels = np.array(training_labels)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:15:32.823594Z","iopub.execute_input":"2022-07-07T12:15:32.824759Z","iopub.status.idle":"2022-07-07T12:15:32.835791Z","shell.execute_reply.started":"2022-07-07T12:15:32.824715Z","shell.execute_reply":"2022-07-07T12:15:32.834359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = tf.keras.Sequential([\n    tf.keras.layers.Embedding(vocab_size, embedding_dim, input_length=max_length),\n    tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(12)),\n    tf.keras.layers.Dense(1, activation = 'sigmoid')\n])\n\nmodel.compile(\n    loss='binary_crossentropy',\n    optimizer='Adamax',\n    metrics=['accuracy', tf.keras.metrics.Precision(), tf.keras.metrics.Recall()]\n)\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:16:46.636642Z","iopub.execute_input":"2022-07-07T12:16:46.637814Z","iopub.status.idle":"2022-07-07T12:16:48.617688Z","shell.execute_reply.started":"2022-07-07T12:16:46.637765Z","shell.execute_reply":"2022-07-07T12:16:48.616199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    training_padded, \n    training_labels, \n    epochs = 10, \n    batch_size = 64,  \n    validation_split=0.2\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:16:55.233995Z","iopub.execute_input":"2022-07-07T12:16:55.234446Z","iopub.status.idle":"2022-07-07T12:17:16.406457Z","shell.execute_reply.started":"2022-07-07T12:16:55.234412Z","shell.execute_reply":"2022-07-07T12:17:16.405072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history[\"accuracy\"], color=\"r\")\nplt.plot(history.history[\"val_accuracy\"], color=\"g\")\nplt.legend([\"Training\", \"Validation\"])\nplt.xlabel(\"epochs\")\nplt.ylabel(\"accuracy\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:17:24.747717Z","iopub.execute_input":"2022-07-07T12:17:24.748142Z","iopub.status.idle":"2022-07-07T12:17:24.967689Z","shell.execute_reply.started":"2022-07-07T12:17:24.748108Z","shell.execute_reply":"2022-07-07T12:17:24.966446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_predict = model.predict(testing_padded).round().astype(int)\n\ntest_predict","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:17:33.656133Z","iopub.execute_input":"2022-07-07T12:17:33.657633Z","iopub.status.idle":"2022-07-07T12:17:35.792901Z","shell.execute_reply.started":"2022-07-07T12:17:33.657584Z","shell.execute_reply":"2022-07-07T12:17:35.791328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({'id':df_test['id'],'target':test_predict.ravel()})\nsubmission.to_csv('submission2.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:18:06.468603Z","iopub.execute_input":"2022-07-07T12:18:06.468972Z","iopub.status.idle":"2022-07-07T12:18:06.485544Z","shell.execute_reply.started":"2022-07-07T12:18:06.468942Z","shell.execute_reply":"2022-07-07T12:18:06.484082Z"},"trusted":true},"execution_count":null,"outputs":[]}]}