{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n!pip install contractions\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport string\nimport re\nimport nltk\nimport subprocess\nfrom nltk.stem import WordNetLemmatizer\nfrom nltk.stem import PorterStemmer\nfrom nltk.stem import SnowballStemmer\nfrom nltk.stem import LancasterStemmer\nimport contractions as ctrs\nimport os\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\ntry:\n    nltk.data.find('wordnet.zip')\nexcept:\n    nltk.download('wordnet', download_dir='/kaggle/working/')\n    command = \"unzip /kaggle/working/corpora/wordnet.zip -d /kaggle/working/corpora\"\n    subprocess.run(command.split())\n    nltk.data.path.append('/kaggle/working/')\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-02T17:06:02.563790Z","iopub.execute_input":"2023-09-02T17:06:02.564059Z","iopub.status.idle":"2023-09-02T17:06:18.189389Z","shell.execute_reply.started":"2023-09-02T17:06:02.564034Z","shell.execute_reply":"2023-09-02T17:06:18.188198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:06:18.191791Z","iopub.execute_input":"2023-09-02T17:06:18.192164Z","iopub.status.idle":"2023-09-02T17:06:20.851588Z","shell.execute_reply.started":"2023-09-02T17:06:18.192127Z","shell.execute_reply":"2023-09-02T17:06:20.850612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:06:20.853259Z","iopub.execute_input":"2023-09-02T17:06:20.853610Z","iopub.status.idle":"2023-09-02T17:06:20.874090Z","shell.execute_reply.started":"2023-09-02T17:06:20.853568Z","shell.execute_reply":"2023-09-02T17:06:20.873220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.drop(columns=['toxic','severe_toxic','obscene','threat','insult','identity_hate'])","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:06:20.876657Z","iopub.execute_input":"2023-09-02T17:06:20.877085Z","iopub.status.idle":"2023-09-02T17:06:20.894771Z","shell.execute_reply.started":"2023-09-02T17:06:20.877052Z","shell.execute_reply":"2023-09-02T17:06:20.893670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def filter_URLS(text):\n    return re.sub(r'https:?//\\S+|www\\.',\"\",text)\ndef filter_emails(text):\n    return re.sub(r'^[a-zA-Z0-9_-]+[@][a-zA-Z]+[.].+',\"\",text)\ndef filter_html(text):\n    return re.sub(r'<.*?>|&([a-z0-9]+|#[0-9]{1,6}|#x[0-9a-f]{1,6});',\"\",text)\ndef remove_punctuations(text):\n    k = string.punctuation\n    return text.translate(str.maketrans(' ',' ',k) )\ndef remove_emojis(text):\n    emoji_pattern = re.compile(\"[\"u\"\\U0001F600-\\U0001F64F\"\n                                   u\"\\U0001F300-\\U0001F5FF\"\n                                   u\"\\U0001F680-\\U0001F6FF\"\n                                   u\"\\U0001F1E0-\\U0001F1FF\"\"]+\", flags=re.UNICODE)\n    return emoji_pattern.sub(r'', text)\ndef remove_non_ascii(text):\n    stripped = (c for c in text if 0 < ord(c) < 127)\n    return ''.join(stripped)\n\ndef remove_chars(text):\n    return re.sub(r'\\n',\" \",text)","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:06:20.896354Z","iopub.execute_input":"2023-09-02T17:06:20.896711Z","iopub.status.idle":"2023-09-02T17:06:20.906558Z","shell.execute_reply.started":"2023-09-02T17:06:20.896679Z","shell.execute_reply":"2023-09-02T17:06:20.905419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['comment_text'] = df_train['comment_text'].apply(lambda x:x.lower())\ndf_train['comment_text'] = df_train['comment_text'].apply(lambda x:ctrs.fix(x))\ndf_train['comment_text'] = df_train['comment_text'].apply(lambda x:filter_URLS(x))\ndf_train['comment_text'] = df_train['comment_text'].apply(lambda x:filter_emails(x))\ndf_train['comment_text'] = df_train['comment_text'].apply(lambda x:filter_html(x))\ndf_train['comment_text'] = df_train['comment_text'].apply(lambda x:remove_punctuations(x))\ndf_train['comment_text'] = df_train['comment_text'].apply(lambda x:remove_emojis(x))\ndf_train['comment_text'] = df_train['comment_text'].apply(lambda x:remove_non_ascii(x))\ndf_train['comment_text'] = df_train['comment_text'].apply(lambda x:remove_chars(x))","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:06:20.908157Z","iopub.execute_input":"2023-09-02T17:06:20.908509Z","iopub.status.idle":"2023-09-02T17:07:03.834468Z","shell.execute_reply.started":"2023-09-02T17:06:20.908478Z","shell.execute_reply":"2023-09-02T17:07:03.833494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:07:03.836278Z","iopub.execute_input":"2023-09-02T17:07:03.836626Z","iopub.status.idle":"2023-09-02T17:07:03.848943Z","shell.execute_reply.started":"2023-09-02T17:07:03.836594Z","shell.execute_reply":"2023-09-02T17:07:03.847726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Text Preprocessing","metadata":{}},{"cell_type":"markdown","source":"**Tokenization**  \n* Tokenization is the process of splitting a text into smaller text items known as tokens.\n* The tokens can be words,subwords,phrases or even characters. \n* Tokens help in building the Vocabulary w.r.t a corpus.\n* Vocabulary is a set of unique tokens\n ","metadata":{"execution":{"iopub.status.busy":"2023-08-30T09:46:53.010711Z","iopub.execute_input":"2023-08-30T09:46:53.011390Z","iopub.status.idle":"2023-08-30T09:46:53.016851Z","shell.execute_reply.started":"2023-08-30T09:46:53.011353Z","shell.execute_reply":"2023-08-30T09:46:53.015529Z"}}},{"cell_type":"code","source":"from nltk.tokenize import word_tokenize\ndf_train[\"tokenized\"] = df_train['comment_text'].apply(word_tokenize)","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:07:03.850736Z","iopub.execute_input":"2023-09-02T17:07:03.851212Z","iopub.status.idle":"2023-09-02T17:09:21.011102Z","shell.execute_reply.started":"2023-09-02T17:07:03.851176Z","shell.execute_reply":"2023-09-02T17:09:21.010096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:09:21.012799Z","iopub.execute_input":"2023-09-02T17:09:21.013222Z","iopub.status.idle":"2023-09-02T17:09:21.028439Z","shell.execute_reply.started":"2023-09-02T17:09:21.013187Z","shell.execute_reply":"2023-09-02T17:09:21.027355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**STOPWORDS**\n* Stop words are essentially filler words present in abundance in a corpus.\n* Example of stop words in English a,the,by,was,is,if.\n* As they don't convey important information in most cases, they are omitted as a preprocessing step\n* Do we remove stop words all the time?\n*     It depends on the use case.\n*     Eg. The food was not very good.\n         *     After removing stop words -> food good \n         *     The remaining text doesn't convey the information that was being projected by the actual statement.\n* Applied on Tokens","metadata":{}},{"cell_type":"code","source":"nltk.download(\"stopwords\")\nfrom nltk.corpus import stopwords\nstop = set(stopwords.words('english'))\ndf_train['no_stopwords'] = df_train['tokenized'].apply(lambda x:[word for word in x if word not in stop])\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:09:21.034340Z","iopub.execute_input":"2023-09-02T17:09:21.034790Z","iopub.status.idle":"2023-09-02T17:09:24.044529Z","shell.execute_reply.started":"2023-09-02T17:09:21.034760Z","shell.execute_reply":"2023-09-02T17:09:24.043619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Stemming**\n* Stemming is the process of reducing a dervied word to it's word stem.\n* The way stemming works is by taking a list of prefix and suffix found in dervied words.\n* The words are then either reduced to it's stem by chopping the prefix or suffix.\n* Stemming is a rule based approach.\n* The following stemmers are in NLTK.\n    * Porter Stemmer\n    * Snowball Stemmer\n    * Lancaster Stemmer\nWill be using the Snowball and Lancaster Stemmers\n    ","metadata":{}},{"cell_type":"code","source":"def p_stemmer(text):\n    p_stem = nltk.PorterStemmer()\n    stems = [p_stem.stem(i) for i in text]\n    return stems\n\ndef s_stemmer(text):\n    s_stem = nltk.SnowballStemmer('english')\n    stems = [s_stem.stem(i) for i in text]\n    return stems\n\ndef l_stemmer(text):\n    l_stem = nltk.LancasterStemmer()\n    stems = [l_stem.stem(i) for i in text]\n    return stems","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:09:24.045834Z","iopub.execute_input":"2023-09-02T17:09:24.046620Z","iopub.status.idle":"2023-09-02T17:09:24.053905Z","shell.execute_reply.started":"2023-09-02T17:09:24.046585Z","shell.execute_reply":"2023-09-02T17:09:24.052881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df_train['p_stemmer']= df_train['no_stopwords'].apply(lambda x:p_stemmer(x))\ndf_train['s_stemmer']= df_train['no_stopwords'].apply(lambda x:s_stemmer(x))\ndf_train['l_stemmer']= df_train['no_stopwords'].apply(lambda x:l_stemmer(x))","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:09:24.055290Z","iopub.execute_input":"2023-09-02T17:09:24.055851Z","iopub.status.idle":"2023-09-02T17:15:20.429583Z","shell.execute_reply.started":"2023-09-02T17:09:24.055817Z","shell.execute_reply":"2023-09-02T17:15:20.428630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:15:20.431167Z","iopub.execute_input":"2023-09-02T17:15:20.431539Z","iopub.status.idle":"2023-09-02T17:15:20.455812Z","shell.execute_reply.started":"2023-09-02T17:15:20.431506Z","shell.execute_reply":"2023-09-02T17:15:20.454748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Lemmitization And POS Tagging**\n* Lemma referes to the dictionary form of a word.\n* The process of Lemmitization is used to reduce inflected words to its root.\n* The way it is different from stemming is that, it takes into consideration the POS tags and the meaning of the word while applying a morphological tranform to convert it to its root form.\n* For example runs, running and ran, after lemmitization gets converted to the run as the root form.","metadata":{}},{"cell_type":"code","source":"def pos_tagged(text):\n    return nltk.pos_tag(text)\n\n# Mapping the nltk pos tags to wordnet format\ndef cvt_to_wnet(token_tag):\n    \n    if token_tag.startswith('N'):\n        return 'n'\n    elif token_tag.startswith('V'):\n        return 'v'\n    elif token_tag.startswith('J'):\n        return 'a'\n    elif token_tag.startswith('R'):\n        return 'r'\n    else:\n        return 'n'","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:15:20.457289Z","iopub.execute_input":"2023-09-02T17:15:20.457753Z","iopub.status.idle":"2023-09-02T17:15:20.467383Z","shell.execute_reply.started":"2023-09-02T17:15:20.457719Z","shell.execute_reply":"2023-09-02T17:15:20.466270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lemmatizer = WordNetLemmatizer()\n\ndef lemmatize_word(text):\n\n    lemma = [lemmatizer.lemmatize(word, tag) for word, tag in text]\n    return lemma","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:15:20.468522Z","iopub.execute_input":"2023-09-02T17:15:20.469063Z","iopub.status.idle":"2023-09-02T17:15:20.477318Z","shell.execute_reply.started":"2023-09-02T17:15:20.469032Z","shell.execute_reply":"2023-09-02T17:15:20.476290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['pos_tagged'] = df_train['no_stopwords'].apply(lambda x:pos_tagged(x))\ndf_train['pos_tagged'] = [[(w,cvt_to_wnet(t)) for w,t in i] for i in df_train['pos_tagged']]\ndf_train['lemma'] = df_train['pos_tagged'].apply(lambda x:lemmatize_word(x))\n","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:15:20.478640Z","iopub.execute_input":"2023-09-02T17:15:20.479768Z","iopub.status.idle":"2023-09-02T17:25:57.024694Z","shell.execute_reply.started":"2023-09-02T17:15:20.479737Z","shell.execute_reply":"2023-09-02T17:25:57.023716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:25:57.026054Z","iopub.execute_input":"2023-09-02T17:25:57.026497Z","iopub.status.idle":"2023-09-02T17:25:57.093062Z","shell.execute_reply.started":"2023-09-02T17:25:57.026463Z","shell.execute_reply":"2023-09-02T17:25:57.091946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Text to Vectors","metadata":{}},{"cell_type":"markdown","source":"\nThere are few approaches to transform the text to vectors:-\n* Count Based\n    *  Bag of words\n    *  TF-IDF\n* Word Embeddings\n    * Word2Vec\n        *  Continuous Bag of words\n        *  Skipgrams\n    * GLove","metadata":{}},{"cell_type":"markdown","source":" **Count Based**","metadata":{}},{"cell_type":"markdown","source":"**Bag of words (B.O.W)**\n* The Bag of words technique converts text into words while keeping a count of the occurence.\n    *     Eg. Kaggle platform is good for data science exploration. Kaggle also has tutorials.\n    *     First step is to remove the stop words from our text\n    *     The sentence post removal of stop words will be -> sentence 1-> Kaggle platform good  data science exploration. \n                                                          -> sentence 2-> Kaggle tutorials are awesome.\n    *     The vocabulary will be [Kaggle,plaform,good,data,science,exploration,.,tutorials,awesome]\n    *     The encoding for sentence 1 [1,1,1,1,1,1,1,0,0]  \n* Drawbacks of Bag of words is long vectors affecting computations.\n* Don't capture relation between words and their meaning in sentence.\n* Out of Vocabulary words are not accounted for.","metadata":{}},{"cell_type":"code","source":"def cv_test(text,nrange):\n    count_vec = CountVectorizer(analyzer='word', ngram_range=(nrange,nrange))\n    vec_form  = count_vec.fit_transform(text).toarray()\n    return count_vec,vec_form","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:25:57.094481Z","iopub.execute_input":"2023-09-02T17:25:57.095232Z","iopub.status.idle":"2023-09-02T17:25:57.100565Z","shell.execute_reply.started":"2023-09-02T17:25:57.095206Z","shell.execute_reply":"2023-09-02T17:25:57.099496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['lemma_text'] = [' '.join(map(str, l)) for l in df_train['lemma']]","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:25:57.101992Z","iopub.execute_input":"2023-09-02T17:25:57.102535Z","iopub.status.idle":"2023-09-02T17:25:58.280690Z","shell.execute_reply.started":"2023-09-02T17:25:57.102503Z","shell.execute_reply":"2023-09-02T17:25:58.279677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_corpus = df_train['lemma_text'][200:205].tolist()\n","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:25:58.282323Z","iopub.execute_input":"2023-09-02T17:25:58.282701Z","iopub.status.idle":"2023-09-02T17:25:58.290151Z","shell.execute_reply.started":"2023-09-02T17:25:58.282652Z","shell.execute_reply":"2023-09-02T17:25:58.289133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_corpus","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:25:58.291433Z","iopub.execute_input":"2023-09-02T17:25:58.292253Z","iopub.status.idle":"2023-09-02T17:25:58.301424Z","shell.execute_reply.started":"2023-09-02T17:25:58.292219Z","shell.execute_reply":"2023-09-02T17:25:58.300330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv_1,emb_1  = cv_test(test_corpus,1)\ncv_2,emb_2  = cv_test(test_corpus,2)\nprint(f\"Embedding {emb_1}\")\nprint(f\"Vocab for uni-grams{cv_1.vocabulary_}\")\nprint(f\"Embeddings {emb_2}\")\nprint(f\"Vocab for bi-grams {cv_2.vocabulary_}\")","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:25:58.302792Z","iopub.execute_input":"2023-09-02T17:25:58.303354Z","iopub.status.idle":"2023-09-02T17:25:58.325033Z","shell.execute_reply.started":"2023-09-02T17:25:58.303323Z","shell.execute_reply":"2023-09-02T17:25:58.323559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**TF-IDF(Term Frequency, Inverse Document frequency)**\n* Is a measure of originality of word by comparing the number of times a word appears in a document multiplied with the number of document it appears in.\n* More weightage is given to the word which appear less than the ones appearing more.\n* Rare words are captured using the term frequency where as the common words are targeted by Inverse Document frequency.\n* Term Frequency is a ratio of no. of representation of words in document to the total no. of words in document.\n* IDF is defined as a ratio of no. of documents/sentences to the no. of documents/sentences containing the word. log((1+n)/(1+doc_freq(word))","metadata":{}},{"cell_type":"code","source":"def tfidf_test(text,nrange):\n    t_vec = TfidfVectorizer(analyzer='word', ngram_range=(nrange,nrange))\n    vec_form  = t_vec.fit_transform(text).toarray()\n    return t_vec,vec_form","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:25:58.326372Z","iopub.execute_input":"2023-09-02T17:25:58.326677Z","iopub.status.idle":"2023-09-02T17:25:58.332557Z","shell.execute_reply.started":"2023-09-02T17:25:58.326633Z","shell.execute_reply":"2023-09-02T17:25:58.331562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t_vector, t_emb = tfidf_test(test_corpus, 1)\nprint(f\"TFIDF_vectorizer for uni-gram {t_vector}\")\nprint(f\"Embedding {t_emb}\")\nprint(f\"Vocab {t_vector.vocabulary_}\")","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:25:58.334296Z","iopub.execute_input":"2023-09-02T17:25:58.334790Z","iopub.status.idle":"2023-09-02T17:25:58.362622Z","shell.execute_reply.started":"2023-09-02T17:25:58.334758Z","shell.execute_reply":"2023-09-02T17:25:58.361732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Word2Vec\n\n* Word2Vec was introduced in 2013 by Google. \n* The main idea of Word2Vec is to have a large vocabulary. \n* Each word in the vocabulary will be a vector. \n* We will use the vector representations to find the similarities between words.\nAlgorithms of Word2Vec\n* Continuous Bag of Words\n  In the CBOW we are trying to predict a centre word given the context words.\n  The model is converting the sentence into pairs of (context word, center word)\n  A window size tells how many words to select for finding the target word. \n  For e.g if the window size = 2 [Hello, I],[pleasant,day]\n  The input dimension for each context word is 1xW. \n  The input get passed to a hidden layer where it's shape is WxN\n  The output shape will be 1XN.\n\n* SkipGram \n  Skipgram model is opposite to the CBOW model. \n  Given the center word here we are trying to predict the context word.\n  For e.g given a sentence the quick brown fox jumps over the lazy dog. Along with a window size of 2\n  for fox the model will try to predict ->[quick,brown],[jumps,over] etc.\n  \n  ","metadata":{}},{"cell_type":"code","source":"import gensim\nfrom gensim.models import KeyedVectors\nw2v = KeyedVectors.load_word2vec_format('/kaggle/input/googlenewsvectors/GoogleNews-vectors-negative300.bin', binary=True,limit=200000)","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:25:58.363810Z","iopub.execute_input":"2023-09-02T17:25:58.364338Z","iopub.status.idle":"2023-09-02T17:26:03.037896Z","shell.execute_reply.started":"2023-09-02T17:25:58.364305Z","shell.execute_reply":"2023-09-02T17:26:03.036857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(w2v.similarity('cat', 'dog'))\nprint(w2v.similarity('cat', 'lions'))","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:26:03.039293Z","iopub.execute_input":"2023-09-02T17:26:03.039657Z","iopub.status.idle":"2023-09-02T17:26:03.051704Z","shell.execute_reply.started":"2023-09-02T17:26:03.039624Z","shell.execute_reply":"2023-09-02T17:26:03.050033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print(w2v[0])\ndef get_vector(tokens,vector,k=300):\n    if len(tokens)<1:\n        return np.zeros(k)\n    \n    vectorized = [vector[word] if word in vector else np.random.rand(k) for word in tokens]\n    lens = len(vectorized)\n    sums = np.sum(vectorized,axis=0)\n    avg = np.divide(sums,lens)\n    return avg\n\n\ndef get_embedding(vectors,text,k=300):\n    embs = text.apply(lambda x:get_vector(x,vectors,k=300))\n    return list(embs)","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:26:03.053135Z","iopub.execute_input":"2023-09-02T17:26:03.053525Z","iopub.status.idle":"2023-09-02T17:26:03.063939Z","shell.execute_reply.started":"2023-09-02T17:26:03.053488Z","shell.execute_reply":"2023-09-02T17:26:03.062732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embeddings_word2vec = get_embedding(w2v, df_train[\"lemma_text\"], k=300)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:26:03.070866Z","iopub.execute_input":"2023-09-02T17:26:03.073180Z","iopub.status.idle":"2023-09-02T17:30:11.476638Z","shell.execute_reply.started":"2023-09-02T17:26:03.073150Z","shell.execute_reply":"2023-09-02T17:30:11.475517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Embedding matrix size\", len(embeddings_word2vec), len(embeddings_word2vec[0]))","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:30:11.478380Z","iopub.execute_input":"2023-09-02T17:30:11.478824Z","iopub.status.idle":"2023-09-02T17:30:11.485489Z","shell.execute_reply.started":"2023-09-02T17:30:11.478784Z","shell.execute_reply":"2023-09-02T17:30:11.484128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"The sentence: \\\"%s\\\" got embedding values: \" % df_train[\"lemma_text\"][20])\nprint(embeddings_word2vec[20])","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:30:11.487125Z","iopub.execute_input":"2023-09-02T17:30:11.487515Z","iopub.status.idle":"2023-09-02T17:30:11.508113Z","shell.execute_reply.started":"2023-09-02T17:30:11.487456Z","shell.execute_reply":"2023-09-02T17:30:11.507058Z"},"trusted":true},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}