{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-22T12:56:19.620818Z","iopub.execute_input":"2022-07-22T12:56:19.621112Z","iopub.status.idle":"2022-07-22T12:56:19.630908Z","shell.execute_reply.started":"2022-07-22T12:56:19.621078Z","shell.execute_reply":"2022-07-22T12:56:19.630270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nfrom pandas.core.common import SettingWithCopyWarning\n\nwarnings.simplefilter(action=\"ignore\", category=SettingWithCopyWarning)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:56:41.327506Z","iopub.execute_input":"2022-07-22T12:56:41.327831Z","iopub.status.idle":"2022-07-22T12:56:41.333680Z","shell.execute_reply.started":"2022-07-22T12:56:41.327799Z","shell.execute_reply":"2022-07-22T12:56:41.332809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#### Data preprocessing","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:57:11.527092Z","iopub.execute_input":"2022-07-22T12:57:11.527412Z","iopub.status.idle":"2022-07-22T12:57:11.531664Z","shell.execute_reply.started":"2022-07-22T12:57:11.527381Z","shell.execute_reply":"2022-07-22T12:57:11.530860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/quora-insincere-questions-classification/train.csv')\ntest_df = pd.read_csv('../input/quora-insincere-questions-classification/test.csv')\nprint(train_df.shape,'train_df')\nprint(test_df.shape,'test_df')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:57:42.768016Z","iopub.execute_input":"2022-07-22T12:57:42.768510Z","iopub.status.idle":"2022-07-22T12:57:48.524438Z","shell.execute_reply.started":"2022-07-22T12:57:42.768460Z","shell.execute_reply":"2022-07-22T12:57:48.523459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_colwidth', None)\n# train_df.iloc[:20,1]\ntarget_1 = train_df[train_df['target']==1]\ntarget_0 = train_df[train_df['target']==0]\n\nprint('count of sincere cases: ',len(target_0))\nprint('count of insincere cases: ',len(target_1))\nprint('% coverage of insincere cases in whole dataset: ',len(target_1)*100/len(train_df) )","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:59:10.132933Z","iopub.execute_input":"2022-07-22T12:59:10.133892Z","iopub.status.idle":"2022-07-22T12:59:10.249598Z","shell.execute_reply.started":"2022-07-22T12:59:10.133852Z","shell.execute_reply":"2022-07-22T12:59:10.248570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_1.sample(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:59:19.228940Z","iopub.execute_input":"2022-07-22T12:59:19.229308Z","iopub.status.idle":"2022-07-22T12:59:19.246229Z","shell.execute_reply.started":"2022-07-22T12:59:19.229270Z","shell.execute_reply":"2022-07-22T12:59:19.244735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:57:56.983775Z","iopub.execute_input":"2022-07-22T12:57:56.984086Z","iopub.status.idle":"2022-07-22T12:57:57.014106Z","shell.execute_reply.started":"2022-07-22T12:57:56.984046Z","shell.execute_reply":"2022-07-22T12:57:57.013302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import string\nstring.punctuation","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:58:01.070244Z","iopub.execute_input":"2022-07-22T12:58:01.070717Z","iopub.status.idle":"2022-07-22T12:58:01.076066Z","shell.execute_reply.started":"2022-07-22T12:58:01.070664Z","shell.execute_reply":"2022-07-22T12:58:01.075324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#defining the function to remove punctuation\ndef remove_punctuation(text):\n    punctuationfree=\"\".join([i for i in text if i not in string.punctuation])\n    return punctuationfree\n\ntarget_1['clean_msg']= target_1['question_text'].apply(lambda x:remove_punctuation(x))\ntarget_1.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:59:23.871890Z","iopub.execute_input":"2022-07-22T12:59:23.872393Z","iopub.status.idle":"2022-07-22T12:59:25.054310Z","shell.execute_reply.started":"2022-07-22T12:59:23.872360Z","shell.execute_reply":"2022-07-22T12:59:25.053402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# target_1.columns.values\ntarget_1['msg_lower']= target_1['clean_msg'].apply(lambda x: x.lower())\ntarget_1.iloc[:5,[3,4]]","metadata":{"execution":{"iopub.status.busy":"2022-07-22T13:00:13.436260Z","iopub.execute_input":"2022-07-22T13:00:13.436583Z","iopub.status.idle":"2022-07-22T13:00:13.500419Z","shell.execute_reply.started":"2022-07-22T13:00:13.436545Z","shell.execute_reply":"2022-07-22T13:00:13.499269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#defining function for tokenization\nimport re\ndef tokenization(text):\n    tokens = re.split('\\W+',text)\n    return tokens\n#applying function to the column\ntarget_1['msg_tokenied']= target_1['msg_lower'].apply(lambda x: tokenization(x))\ntarget_1.iloc[:3,-1:]\n# text = 'if blacks support school choicey vote republican'\n# tokens = re.split('\\W+',text)\n# tokens\n# import nltk\n# word_data = 'if blacks support school choicey vote republican'\n# nltk_tokens = nltk.word_tokenize(word_data)\n# print (nltk_tokens)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T13:00:39.978789Z","iopub.execute_input":"2022-07-22T13:00:39.979103Z","iopub.status.idle":"2022-07-22T13:00:41.110710Z","shell.execute_reply.started":"2022-07-22T13:00:39.979070Z","shell.execute_reply":"2022-07-22T13:00:41.109701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#importing nlp library\nimport nltk\n#Stop words present in the library\nstopwords = nltk.corpus.stopwords.words('english')\nstopwords[0:10]","metadata":{"execution":{"iopub.status.busy":"2022-07-22T13:01:11.406671Z","iopub.execute_input":"2022-07-22T13:01:11.407131Z","iopub.status.idle":"2022-07-22T13:01:11.418428Z","shell.execute_reply.started":"2022-07-22T13:01:11.407083Z","shell.execute_reply":"2022-07-22T13:01:11.417569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#defining the function to remove stopwords from tokenized text\ndef remove_stopwords(text):\n    output= [i for i in text if i not in stopwords]\n    return output\n\ntarget_1['stpw_remov'] = target_1['msg_tokenied'].apply(lambda x : remove_stopwords(x))\ntarget_1.iloc[:3,-2:]\n# target_1.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T13:01:13.809671Z","iopub.execute_input":"2022-07-22T13:01:13.810389Z","iopub.status.idle":"2022-07-22T13:01:16.545538Z","shell.execute_reply.started":"2022-07-22T13:01:13.810338Z","shell.execute_reply":"2022-07-22T13:01:16.543277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from nltk.stem.porter import PorterStemmer\nporter_stemmer = PorterStemmer()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T13:01:34.445008Z","iopub.execute_input":"2022-07-22T13:01:34.445346Z","iopub.status.idle":"2022-07-22T13:01:34.450702Z","shell.execute_reply.started":"2022-07-22T13:01:34.445314Z","shell.execute_reply":"2022-07-22T13:01:34.449618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def stemming(text):\n    stem_text = [porter_stemmer.stem(word) for word in text]\n    return stem_text\n\ntarget_1['msg_stemmed']=target_1['stpw_remov'].apply(lambda x: stemming(x))\ntarget_1.iloc[:3,-2:]","metadata":{"execution":{"iopub.status.busy":"2022-07-22T13:03:01.520501Z","iopub.execute_input":"2022-07-22T13:03:01.520778Z","iopub.status.idle":"2022-07-22T13:03:24.342709Z","shell.execute_reply.started":"2022-07-22T13:03:01.520751Z","shell.execute_reply":"2022-07-22T13:03:24.341779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from nltk.stem import WordNetLemmatizer\nwordnet_lemmatizer = WordNetLemmatizer()\n\ndef lemmatizer(text):\n    lemm_text = [wordnet_lemmatizer.lemmatize(word) for word in text]\n    return lemm_text\n\ntarget_1['msg_lemmatized']=target_1['stpw_remov'].apply(lambda x:lemmatizer(x))\ntarget_1.iloc[:3,-3:]","metadata":{"execution":{"iopub.status.busy":"2022-07-22T13:03:31.593626Z","iopub.execute_input":"2022-07-22T13:03:31.593984Z","iopub.status.idle":"2022-07-22T13:03:38.500112Z","shell.execute_reply.started":"2022-07-22T13:03:31.593948Z","shell.execute_reply":"2022-07-22T13:03:38.499039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df=pd.read_csv('../input/quora-insincere-questions-classification/sample_submission.csv')\nprint(sub_df.shape,'sub_df')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T13:03:45.232832Z","iopub.execute_input":"2022-07-22T13:03:45.233168Z","iopub.status.idle":"2022-07-22T13:03:45.637529Z","shell.execute_reply.started":"2022-07-22T13:03:45.233131Z","shell.execute_reply":"2022-07-22T13:03:45.636283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom wordcloud import WordCloud\nfrom wordcloud import ImageColorGenerator\nfrom wordcloud import STOPWORDS\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T13:03:46.989115Z","iopub.execute_input":"2022-07-22T13:03:46.989464Z","iopub.status.idle":"2022-07-22T13:03:47.032804Z","shell.execute_reply.started":"2022-07-22T13:03:46.989388Z","shell.execute_reply":"2022-07-22T13:03:47.031632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text = \" \".join(i for i in target_1.question_text)\nstopwords = set(STOPWORDS)\nwordcloud = WordCloud(stopwords=stopwords, background_color=\"white\").generate(text)\nplt.figure( figsize=(15,10))\nplt.imshow(wordcloud, interpolation='bilinear')\nplt.axis(\"off\")\nplt.show()\n\n# text = target_1['question_text'].values \n# # print(text)\n# wordcloud = WordCloud().generate(str(text))\n\n# plt.imshow(wordcloud)\n# plt.axis(\"off\")\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T13:04:26.000402Z","iopub.execute_input":"2022-07-22T13:04:26.000857Z","iopub.status.idle":"2022-07-22T13:04:32.799132Z","shell.execute_reply.started":"2022-07-22T13:04:26.000818Z","shell.execute_reply":"2022-07-22T13:04:32.798242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# EDA: the conclusion --\n# We gained the following knowledge by doing Exploratory Data Knowledge:\n\n# There are no null rows, and no duplicate rows\n# We have hell-a-lot of data imbalance! (~93% of target=0 and ~7% of target=1)\n# There is bias, ie, we can see community specific and location specific terms skewed towards a category\n# Both unigram and bigram analysis shows the abundance of Stop Words (Spoiler alert: we're not doing anything to treat it😉)\n# We are unable to draw a clear relation between question length, average word length and target class (they come in all shapes and sizes 🏈⚽)\n# We have a few HTML tags, and a few HTTP URLs\n# We only have ©orporate emoticons™ in our dataset®\n# The dataset is enriched with punctuations, keeping them could hopefully contribute to the knowledge mining process❕‼","metadata":{"execution":{"iopub.status.busy":"2022-07-22T13:06:55.275647Z","iopub.execute_input":"2022-07-22T13:06:55.275951Z","iopub.status.idle":"2022-07-22T13:06:55.281527Z","shell.execute_reply.started":"2022-07-22T13:06:55.275921Z","shell.execute_reply":"2022-07-22T13:06:55.280470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Don't use standard preprocessing steps like stemming or stopword removal when you have pre-trained embeddings\n# Some of you might used standard preprocessing steps when doing word count\n# based feature extraction (e.g. TFIDF) such as removing stopwords, stemming etc. \n# The reason is simple: You loose valuable information, which would help your NN to figure things out.\n\n\n# Get your vocabulary as close to the embeddings as possible","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom tqdm import tqdm\ntqdm.pandas()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:54:11.366984Z","iopub.execute_input":"2022-08-04T13:54:11.367313Z","iopub.status.idle":"2022-08-04T13:54:11.375978Z","shell.execute_reply.started":"2022-08-04T13:54:11.367282Z","shell.execute_reply":"2022-08-04T13:54:11.375107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:39:37.670575Z","iopub.execute_input":"2022-08-04T13:39:37.670865Z","iopub.status.idle":"2022-08-04T13:39:38.802956Z","shell.execute_reply.started":"2022-08-04T13:39:37.670836Z","shell.execute_reply":"2022-08-04T13:39:38.801587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# /kaggle/input/quora-insincere-questions-classificationcd ..\n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:49:03.722516Z","iopub.execute_input":"2022-08-04T13:49:03.723728Z","iopub.status.idle":"2022-08-04T13:49:03.729420Z","shell.execute_reply.started":"2022-08-04T13:49:03.723641Z","shell.execute_reply":"2022-08-04T13:49:03.728279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!unzip -q /kaggle/input/quora-insincere-questions-classification/embeddings.zip","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:39:46.998156Z","iopub.execute_input":"2022-08-04T13:39:46.998528Z","iopub.status.idle":"2022-08-04T13:43:30.432325Z","shell.execute_reply.started":"2022-08-04T13:39:46.998492Z","shell.execute_reply":"2022-08-04T13:43:30.428932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cd GoogleNews-vectors-negative300","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:51:22.626245Z","iopub.execute_input":"2022-08-04T13:51:22.626610Z","iopub.status.idle":"2022-08-04T13:51:22.638267Z","shell.execute_reply.started":"2022-08-04T13:51:22.626576Z","shell.execute_reply":"2022-08-04T13:51:22.636818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:51:32.640284Z","iopub.execute_input":"2022-08-04T13:51:32.640627Z","iopub.status.idle":"2022-08-04T13:51:33.789465Z","shell.execute_reply.started":"2022-08-04T13:51:32.640593Z","shell.execute_reply":"2022-08-04T13:51:33.788445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cd input\n# !wget https://s3.amazonaws.com/dl4j-distribution/GoogleNews-vectors-negative300.bin.gz","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:50:12.953082Z","iopub.execute_input":"2022-08-04T11:50:12.953369Z","iopub.status.idle":"2022-08-04T11:50:14.219912Z","shell.execute_reply.started":"2022-08-04T11:50:12.953341Z","shell.execute_reply":"2022-08-04T11:50:14.218632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:51:36.686951Z","iopub.execute_input":"2022-08-04T11:51:36.687279Z","iopub.status.idle":"2022-08-04T11:51:37.831437Z","shell.execute_reply.started":"2022-08-04T11:51:36.687246Z","shell.execute_reply":"2022-08-04T11:51:37.830585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pwd\n# cd ..\n# !wget http://nlp.stanford.edu/data/glove.6B.zip\n# !gzip -d GoogleNews-vectors-negative300.bin.gz","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:55:29.231979Z","iopub.execute_input":"2022-08-04T13:55:29.232334Z","iopub.status.idle":"2022-08-04T13:55:29.237685Z","shell.execute_reply.started":"2022-08-04T13:55:29.232301Z","shell.execute_reply":"2022-08-04T13:55:29.236233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from gensim.models import KeyedVectors\n\nnews_path = '/kaggle/working/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin'\nembeddings_index = KeyedVectors.load_word2vec_format(news_path, binary=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:51:47.978216Z","iopub.execute_input":"2022-08-04T13:51:47.978610Z","iopub.status.idle":"2022-08-04T13:52:29.038886Z","shell.execute_reply.started":"2022-08-04T13:51:47.978573Z","shell.execute_reply":"2022-08-04T13:52:29.037526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import operator \n\ndef check_coverage(vocab,embeddings_index):\n    a = {}\n    oov = {}\n    k = 0\n    i = 0\n    for word in tqdm(vocab):\n        try:\n            a[word] = embeddings_index[word]\n            k += vocab[word]\n        except:\n\n            oov[word] = vocab[word]\n            i += vocab[word]\n            pass\n\n    print('Found embeddings for {:.2%} of vocab'.format(len(a) / len(vocab)))\n    print('Found embeddings for  {:.2%} of all text'.format(k / (k + i)))\n    sorted_x = sorted(oov.items(), key=operator.itemgetter(1))[::-1]\n\n    return sorted_x","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:53:39.617650Z","iopub.execute_input":"2022-08-04T13:53:39.618002Z","iopub.status.idle":"2022-08-04T13:53:39.627068Z","shell.execute_reply.started":"2022-08-04T13:53:39.617966Z","shell.execute_reply":"2022-08-04T13:53:39.625806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cd input\n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:56:24.439110Z","iopub.execute_input":"2022-08-04T13:56:24.439509Z","iopub.status.idle":"2022-08-04T13:56:24.447856Z","shell.execute_reply.started":"2022-08-04T13:56:24.439467Z","shell.execute_reply":"2022-08-04T13:56:24.446794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/quora-insincere-questions-classification/train.csv')\ntest = pd.read_csv('../input/quora-insincere-questions-classification/test.csv')\nprint(train.shape,'train_df') \nprint(test.shape,'test_df')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:56:27.240208Z","iopub.execute_input":"2022-08-04T13:56:27.240578Z","iopub.status.idle":"2022-08-04T13:56:33.410430Z","shell.execute_reply.started":"2022-08-04T13:56:27.240545Z","shell.execute_reply":"2022-08-04T13:56:33.409376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# To track our training vocabulary, which goes through all our text and counts the occurance of the contained words\ndef build_vocab(sentences, verbose =  True):\n    \"\"\"\n    :param sentences: list of list of words\n    :return: dictionary of words and their count\n    \"\"\"\n    vocab = {}\n    for sentence in tqdm(sentences, disable = (not verbose)):\n        for word in sentence:\n            try:\n                vocab[word] += 1\n            except KeyError:\n                vocab[word] = 1\n    return vocab","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:56:37.613449Z","iopub.execute_input":"2022-08-04T13:56:37.613807Z","iopub.status.idle":"2022-08-04T13:56:37.621559Z","shell.execute_reply.started":"2022-08-04T13:56:37.613774Z","shell.execute_reply":"2022-08-04T13:56:37.620428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sentences = train[\"question_text\"].progress_apply(lambda x: x.split()).values\nvocab = build_vocab(sentences)\nprint({k: vocab[k] for k in list(vocab)[:5]})","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:56:44.063861Z","iopub.execute_input":"2022-08-04T13:56:44.064225Z","iopub.status.idle":"2022-08-04T13:56:58.101470Z","shell.execute_reply.started":"2022-08-04T13:56:44.064186Z","shell.execute_reply":"2022-08-04T13:56:58.100204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oov = check_coverage(vocab,embeddings_index)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:57:11.428698Z","iopub.execute_input":"2022-08-04T13:57:11.429627Z","iopub.status.idle":"2022-08-04T13:57:13.427138Z","shell.execute_reply.started":"2022-08-04T13:57:11.429551Z","shell.execute_reply":"2022-08-04T13:57:13.425943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oov[:10]","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:57:26.154734Z","iopub.execute_input":"2022-08-04T13:57:26.155060Z","iopub.status.idle":"2022-08-04T13:57:26.176254Z","shell.execute_reply.started":"2022-08-04T13:57:26.155021Z","shell.execute_reply":"2022-08-04T13:57:26.175212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'?' in embeddings_index","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:57:42.252380Z","iopub.execute_input":"2022-08-04T13:57:42.252773Z","iopub.status.idle":"2022-08-04T13:57:42.260767Z","shell.execute_reply.started":"2022-08-04T13:57:42.252736Z","shell.execute_reply":"2022-08-04T13:57:42.259767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'&' in embeddings_index","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:57:58.208902Z","iopub.execute_input":"2022-08-04T13:57:58.209242Z","iopub.status.idle":"2022-08-04T13:57:58.217389Z","shell.execute_reply.started":"2022-08-04T13:57:58.209210Z","shell.execute_reply":"2022-08-04T13:57:58.216404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_text(x):\n\n    x = str(x)\n    for punct in \"/-'\":\n        x = x.replace(punct, ' ')\n    for punct in '&':\n        x = x.replace(punct, f' {punct} ')\n    for punct in '?!.,\"#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~' + '“”’':\n        x = x.replace(punct, '')\n    return x","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:58:09.279557Z","iopub.execute_input":"2022-08-04T13:58:09.279967Z","iopub.status.idle":"2022-08-04T13:58:09.287414Z","shell.execute_reply.started":"2022-08-04T13:58:09.279928Z","shell.execute_reply":"2022-08-04T13:58:09.286057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"question_text\"] = train[\"question_text\"].progress_apply(lambda x: clean_text(x))\nsentences = train[\"question_text\"].apply(lambda x: x.split())\nvocab = build_vocab(sentences)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:58:17.371486Z","iopub.execute_input":"2022-08-04T13:58:17.372308Z","iopub.status.idle":"2022-08-04T13:58:43.329106Z","shell.execute_reply.started":"2022-08-04T13:58:17.372251Z","shell.execute_reply":"2022-08-04T13:58:43.328185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oov = check_coverage(vocab,embeddings_index)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:58:52.140939Z","iopub.execute_input":"2022-08-04T13:58:52.141323Z","iopub.status.idle":"2022-08-04T13:58:53.568751Z","shell.execute_reply.started":"2022-08-04T13:58:52.141285Z","shell.execute_reply":"2022-08-04T13:58:53.567796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oov[:10]","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:58:56.567192Z","iopub.execute_input":"2022-08-04T13:58:56.567701Z","iopub.status.idle":"2022-08-04T13:58:56.575095Z","shell.execute_reply.started":"2022-08-04T13:58:56.567640Z","shell.execute_reply":"2022-08-04T13:58:56.574226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(10):\n    print(embeddings_index.index_to_key[i])","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:59:15.038684Z","iopub.execute_input":"2022-08-04T13:59:15.039631Z","iopub.status.idle":"2022-08-04T13:59:15.046607Z","shell.execute_reply.started":"2022-08-04T13:59:15.039575Z","shell.execute_reply":"2022-08-04T13:59:15.045678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\n\ndef clean_numbers(x):\n\n    x = re.sub('[0-9]{5,}', '#####', x)\n    x = re.sub('[0-9]{4}', '####', x)\n    x = re.sub('[0-9]{3}', '###', x)\n    x = re.sub('[0-9]{2}', '##', x)\n    return x","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:59:30.041221Z","iopub.execute_input":"2022-08-04T13:59:30.041620Z","iopub.status.idle":"2022-08-04T13:59:30.049622Z","shell.execute_reply.started":"2022-08-04T13:59:30.041574Z","shell.execute_reply":"2022-08-04T13:59:30.048506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"question_text\"] = train[\"question_text\"].progress_apply(lambda x: clean_numbers(x))\nsentences = train[\"question_text\"].progress_apply(lambda x: x.split())\nvocab = build_vocab(sentences)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:59:41.351161Z","iopub.execute_input":"2022-08-04T13:59:41.351509Z","iopub.status.idle":"2022-08-04T14:00:10.610209Z","shell.execute_reply.started":"2022-08-04T13:59:41.351473Z","shell.execute_reply":"2022-08-04T14:00:10.609193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oov = check_coverage(vocab,embeddings_index)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:00:13.565335Z","iopub.execute_input":"2022-08-04T14:00:13.565696Z","iopub.status.idle":"2022-08-04T14:00:14.762675Z","shell.execute_reply.started":"2022-08-04T14:00:13.565661Z","shell.execute_reply":"2022-08-04T14:00:14.761043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oov[:20]","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:00:26.986419Z","iopub.execute_input":"2022-08-04T14:00:26.989811Z","iopub.status.idle":"2022-08-04T14:00:27.014520Z","shell.execute_reply.started":"2022-08-04T14:00:26.989518Z","shell.execute_reply":"2022-08-04T14:00:27.011968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def _get_mispell(mispell_dict):\n    mispell_re = re.compile('(%s)' % '|'.join(mispell_dict.keys()))\n    return mispell_dict, mispell_re\n\n\nmispell_dict = {'colour':'color',\n                'centre':'center',\n                'didnt':'did not',\n                'doesnt':'does not',\n                'isnt':'is not',\n                'shouldnt':'should not',\n                'favourite':'favorite',\n                'travelling':'traveling',\n                'counselling':'counseling',\n                'theatre':'theater',\n                'cancelled':'canceled',\n                'labour':'labor',\n                'organisation':'organization',\n                'wwii':'world war 2',\n                'citicise':'criticize',\n                'instagram': 'social medium',\n                'whatsapp': 'social medium',\n                'snapchat': 'social medium'\n\n                }\nmispellings, mispellings_re = _get_mispell(mispell_dict)\n\ndef replace_typical_misspell(text):\n    def replace(match):\n        return mispellings[match.group(0)]\n\n    return mispellings_re.sub(replace, text)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:00:52.733605Z","iopub.execute_input":"2022-08-04T14:00:52.733918Z","iopub.status.idle":"2022-08-04T14:00:52.743498Z","shell.execute_reply.started":"2022-08-04T14:00:52.733887Z","shell.execute_reply":"2022-08-04T14:00:52.742525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"question_text\"] = train[\"question_text\"].progress_apply(lambda x: replace_typical_misspell(x))\nsentences = train[\"question_text\"].progress_apply(lambda x: x.split())\nto_remove = ['a','to','of','and']\nsentences = [[word for word in sentence if not word in to_remove] for sentence in tqdm(sentences)]\nvocab = build_vocab(sentences)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:01:10.471666Z","iopub.execute_input":"2022-08-04T14:01:10.472468Z","iopub.status.idle":"2022-08-04T14:01:39.157418Z","shell.execute_reply.started":"2022-08-04T14:01:10.472420Z","shell.execute_reply":"2022-08-04T14:01:39.156006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oov = check_coverage(vocab,embeddings_index)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:01:45.848901Z","iopub.execute_input":"2022-08-04T14:01:45.849257Z","iopub.status.idle":"2022-08-04T14:01:46.971897Z","shell.execute_reply.started":"2022-08-04T14:01:45.849223Z","shell.execute_reply":"2022-08-04T14:01:46.970800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oov[:20]","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:02:47.868141Z","iopub.execute_input":"2022-08-04T14:02:47.868504Z","iopub.status.idle":"2022-08-04T14:02:47.877380Z","shell.execute_reply.started":"2022-08-04T14:02:47.868469Z","shell.execute_reply":"2022-08-04T14:02:47.876391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!wget http://nlp.stanford.edu/data/glove.6B.zip\n!unzip -q glove.6B.zip","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_to_glove_file = \"glove.6B.100d.txt\"\n\nembeddings_index = {}\nwith open(path_to_glove_file) as f:\n    for line in f:\n        word, coefs = line.split(maxsplit=1)\n        coefs = np.fromstring(coefs, \"f\", sep=\" \")\n        embeddings_index[word] = coefs\n\nprint(\"Found %s word vectors.\" % len(embeddings_index))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import operator \n\ndef check_coverage(vocab,embeddings_index):\n    a = {}\n    oov = {}\n    k = 0\n    i = 0\n    for word in tqdm(vocab):\n        try:\n            a[word] = embeddings_index[word]\n            k += vocab[word]\n        except:\n\n            oov[word] = vocab[word]\n            i += vocab[word]\n            pass\n\n    print('Found embeddings for {:.2%} of vocab'.format(len(a) / len(vocab)))\n    print('Found embeddings for  {:.2%} of all text'.format(k / (k + i)))\n    sorted_x = sorted(oov.items(), key=operator.itemgetter(1))[::-1]\n\n    return sorted_x\n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:07:14.368760Z","iopub.execute_input":"2022-07-28T13:07:14.369173Z","iopub.status.idle":"2022-07-28T13:07:14.378079Z","shell.execute_reply.started":"2022-07-28T13:07:14.369130Z","shell.execute_reply":"2022-07-28T13:07:14.377249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oov = check_coverage(vocab,embeddings_index)\noov[:10]","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:07:32.124666Z","iopub.execute_input":"2022-07-28T13:07:32.125014Z","iopub.status.idle":"2022-07-28T13:07:32.971903Z","shell.execute_reply.started":"2022-07-28T13:07:32.124981Z","shell.execute_reply":"2022-07-28T13:07:32.971085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'%' in embeddings_index","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:09:10.085302Z","iopub.execute_input":"2022-07-28T13:09:10.085642Z","iopub.status.idle":"2022-07-28T13:09:10.091988Z","shell.execute_reply.started":"2022-07-28T13:09:10.085609Z","shell.execute_reply":"2022-07-28T13:09:10.091089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_text(x):\n\n    x = str(x)\n    for punct in \"/-'\":\n        x = x.replace(punct, ' ')\n    for punct in '&':\n        x = x.replace(punct, f' {punct} ')\n    for punct in '?':\n        x = x.replace(punct, f' {punct} ')\n    for punct in '!':\n        x = x.replace(punct, f' {punct} ')\n    for punct in '$':\n        x = x.replace(punct, f' {punct} ')\n    for punct in '%':\n        x = x.replace(punct, f' {punct} ')\n        \n    for punct in '.,\"#\\'()*+-/:;<=>@[\\\\]^_`{|}~' + '“”’':\n        x = x.replace(punct, '')\n    return x","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:09:27.326668Z","iopub.execute_input":"2022-07-28T13:09:27.326961Z","iopub.status.idle":"2022-07-28T13:09:27.335256Z","shell.execute_reply.started":"2022-07-28T13:09:27.326930Z","shell.execute_reply":"2022-07-28T13:09:27.334206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"question_text\"] = train[\"question_text\"].progress_apply(lambda x: clean_text(x))\nsentences = train[\"question_text\"].apply(lambda x: x.split())\nvocab = build_vocab(sentences)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:09:37.661570Z","iopub.execute_input":"2022-07-28T13:09:37.661869Z","iopub.status.idle":"2022-07-28T13:10:05.872416Z","shell.execute_reply.started":"2022-07-28T13:09:37.661824Z","shell.execute_reply":"2022-07-28T13:10:05.871515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oov = check_coverage(vocab,embeddings_index)\noov[:10]","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:10:11.660604Z","iopub.execute_input":"2022-07-28T13:10:11.660944Z","iopub.status.idle":"2022-07-28T13:10:12.509801Z","shell.execute_reply.started":"2022-07-28T13:10:11.660909Z","shell.execute_reply":"2022-07-28T13:10:12.509186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list(embeddings_index)[:20]\n# embeddings_index['wwii']","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:18:02.341813Z","iopub.execute_input":"2022-07-28T13:18:02.342109Z","iopub.status.idle":"2022-07-28T13:18:02.361251Z","shell.execute_reply.started":"2022-07-28T13:18:02.342079Z","shell.execute_reply":"2022-07-28T13:18:02.360407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:18:53.655910Z","iopub.execute_input":"2022-07-28T13:18:53.656199Z","iopub.status.idle":"2022-07-28T13:18:53.661641Z","shell.execute_reply.started":"2022-07-28T13:18:53.656169Z","shell.execute_reply":"2022-07-28T13:18:53.660303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def _get_mispell(mispell_dict):\n    mispell_re = re.compile('(%s)' % '|'.join(mispell_dict.keys()))\n    return mispell_dict, mispell_re\n\n\nmispell_dict = {'colour':'color',\n                'centre':'center',\n                'didnt':'did not',\n                'doesnt':'does not',\n                'isnt':'is not',\n                'shouldnt':'should not',\n                'favourite':'favorite',\n                'travelling':'traveling',\n                'counselling':'counseling',\n                'theatre':'theater',\n                'cancelled':'canceled',\n                'labour':'labor',\n                'organisation':'organization',\n                'wwii':'world war 2',\n                'citicise':'criticize',\n                'instagram': 'social medium',\n                'whatsapp': 'social medium',\n                'snapchat': 'social medium'\n\n                }\nmispellings, mispellings_re = _get_mispell(mispell_dict)\n\ndef replace_typical_misspell(text):\n    def replace(match):\n        return mispellings[match.group(0)]\n\n    return mispellings_re.sub(replace, text)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:18:56.803538Z","iopub.execute_input":"2022-07-28T13:18:56.804542Z","iopub.status.idle":"2022-07-28T13:18:56.815593Z","shell.execute_reply.started":"2022-07-28T13:18:56.804494Z","shell.execute_reply":"2022-07-28T13:18:56.814617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"question_text\"] = train[\"question_text\"].progress_apply(lambda x: replace_typical_misspell(x))\nsentences = train[\"question_text\"].progress_apply(lambda x: x.split())\nto_remove = ['a','to','of','and']\nsentences = [[word for word in sentence if not word in to_remove] for sentence in tqdm(sentences)]\nvocab = build_vocab(sentences)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:19:12.617894Z","iopub.execute_input":"2022-07-28T13:19:12.618322Z","iopub.status.idle":"2022-07-28T13:19:40.573451Z","shell.execute_reply.started":"2022-07-28T13:19:12.618290Z","shell.execute_reply":"2022-07-28T13:19:40.572473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oov = check_coverage(vocab,embeddings_index)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:20:31.048970Z","iopub.execute_input":"2022-07-28T13:20:31.049283Z","iopub.status.idle":"2022-07-28T13:20:31.807757Z","shell.execute_reply.started":"2022-07-28T13:20:31.049251Z","shell.execute_reply":"2022-07-28T13:20:31.806782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oov[:20]","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:20:45.499905Z","iopub.execute_input":"2022-07-28T13:20:45.500260Z","iopub.status.idle":"2022-07-28T13:20:45.508140Z","shell.execute_reply.started":"2022-07-28T13:20:45.500227Z","shell.execute_reply":"2022-07-28T13:20:45.507407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.layers import Embedding\n\nembedding_layer = Embedding(\n    num_tokens,\n    embedding_dim,\n    embeddings_initializer=keras.initializers.Constant(embedding_matrix),\n    trainable=False,\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:22:19.182382Z","iopub.execute_input":"2022-07-28T13:22:19.183726Z","iopub.status.idle":"2022-07-28T13:22:19.736513Z","shell.execute_reply.started":"2022-07-28T13:22:19.183678Z","shell.execute_reply":"2022-07-28T13:22:19.735310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}