{"cells":[{"metadata":{"_uuid":"710ed17d0c57bd287be0ee3b2782a53a54510561"},"cell_type":"markdown","source":"## IMPORTS "},{"metadata":{"_uuid":"585eec8b444e194eda5a141d2a2316547235ffdf"},"cell_type":"markdown","source":"In This kernel, I will try to use the normal conventional methods on the quora dataset namely:\n\n- TFIDF,\n- CountVectorizer, \n- HashVectorizer,\n- Word embeddings. \n\nTo get an understanding of these I have created a notebook at: https://mlwhiz.com/blog/2019/02/08/deeplearning_nlp_conventional_methods/\n\nDo have a look at the blog post too."},{"metadata":{"_uuid":"abb7e3c30b8a412a50c6b451c49939e3cf4bc11b","scrolled":true,"trusted":true},"cell_type":"code","source":"import random\nimport copy\nimport time\nimport pandas as pd\nimport numpy as np\nimport gc\nimport re\nimport torch\nfrom torchtext import data\n#import spacy\nfrom tqdm import tqdm_notebook, tnrange\nfrom tqdm.auto import tqdm\nfrom sklearn.model_selection import train_test_split\ntqdm.pandas(desc='Progress')\nfrom collections import Counter\nfrom textblob import TextBlob\nfrom nltk import word_tokenize\nfrom nltk.stem import WordNetLemmatizer\nfrom nltk.tokenize.toktok import ToktokTokenizer\nfrom sklearn.feature_extraction.text import CountVectorizer,TfidfVectorizer, HashingVectorizer\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import f1_score\n\nimport os \nimport nltk\nfrom sklearn import preprocessing, decomposition, model_selection, metrics, pipeline\nfrom sklearn.svm import SVC\nfrom sklearn.naive_bayes import MultinomialNB\n\n# cross validation and metrics\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import f1_score\nfrom unidecode import unidecode\n\nfrom sklearn.preprocessing import StandardScaler\nfrom textblob import TextBlob\nfrom multiprocessing import  Pool\nfrom functools import partial\nimport numpy as np\nfrom sklearn.decomposition import PCA\nfrom sklearn.decomposition import TruncatedSVD\nfrom sklearn.svm import LinearSVC\nimport lightgbm as lgb","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9a4ff5590a6f152dc1bec5aeca79aef10218f7de"},"cell_type":"markdown","source":"### Basic Parameters"},{"metadata":{"_uuid":"deee49df5ca1c4413f71677939e26aa1ff784e44","scrolled":true,"trusted":true},"cell_type":"code","source":"SEED = 1029","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ea10c8e218a1280faa9802bcb7f1117c89ec96f9"},"cell_type":"markdown","source":"## LOAD PROCESSED TRAINING DATA FROM DISK"},{"metadata":{"_uuid":"abeab4c80d6829cf2eae706bfa7929e2871af81f","scrolled":true,"trusted":true},"cell_type":"code","source":"# Some preprocesssing that will be common to all the text classification methods you will see. \n\n# Remove punctuations:\npuncts = [',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\', '•',  '~', '@', '£', \n '·', '_', '{', '}', '©', '^', '®', '`',  '<', '→', '°', '€', '™', '›',  '♥', '←', '×', '§', '″', '′', 'Â', '█', '½', 'à', '…', \n '“', '★', '”', '–', '●', 'â', '►', '−', '¢', '²', '¬', '░', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥', '▓', '—', '‹', '─', \n '▒', '：', '¼', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', 'é', '¯', '♦', '¤', '▲', 'è', '¸', '¾', 'Ã', '⋅', '‘', '∞', \n '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '³', '・', '╦', '╣', '╔', '╗', '▬', '❤', 'ï', 'Ø', '¹', '≤', '‡', '√', ]\n\ndef clean_text(x):\n    x = str(x)\n    for punct in puncts:\n        if punct in x:\n            x = x.replace(punct, ' ')\n    return x\n\n# We won't clean numbers in conventional methods case since we might get extra info from bigrams like 5 mins or 30 mins\ndef clean_numbers(x):\n    if bool(re.search(r'\\d', x)):\n        x = re.sub('[0-9]{5,}', '#####', x)\n        x = re.sub('[0-9]{4}', '####', x)\n        x = re.sub('[0-9]{3}', '###', x)\n        x = re.sub('[0-9]{2}', '##', x)\n    return x\n\n# Remove Misspell:\nmispell_dict = {'colour': 'color', 'centre': 'center', 'favourite': 'favorite', 'travelling': 'traveling', 'counselling': 'counseling', 'theatre': 'theater', 'cancelled': 'canceled', 'labour': 'labor', 'organisation': 'organization', 'wwii': 'world war 2', 'citicise': 'criticize', 'youtu ': 'youtube ', 'Qoura': 'Quora', 'sallary': 'salary', 'Whta': 'What', 'narcisist': 'narcissist', 'howdo': 'how do', 'whatare': 'what are', 'howcan': 'how can', 'howmuch': 'how much', 'howmany': 'how many', 'whydo': 'why do', 'doI': 'do I', 'theBest': 'the best', 'howdoes': 'how does', 'mastrubation': 'masturbation', 'mastrubate': 'masturbate', \"mastrubating\": 'masturbating', 'pennis': 'penis', 'Etherium': 'Ethereum', 'narcissit': 'narcissist', 'bigdata': 'big data', '2k17': '2017', '2k18': '2018', 'qouta': 'quota', 'exboyfriend': 'ex boyfriend', 'airhostess': 'air hostess', \"whst\": 'what', 'watsapp': 'whatsapp', 'demonitisation': 'demonetization', 'demonitization': 'demonetization', 'demonetisation': 'demonetization'}\n\ndef _get_mispell(mispell_dict):\n    mispell_re = re.compile('(%s)' % '|'.join(mispell_dict.keys()))\n    return mispell_dict, mispell_re\n\nmispellings, mispellings_re = _get_mispell(mispell_dict)\ndef replace_typical_misspell(text):\n    def replace(match):\n        return mispellings[match.group(0)]\n    return mispellings_re.sub(replace, text)\n\n# remove stopwords:\nstopword_list = nltk.corpus.stopwords.words('english')\ndef remove_stopwords(text, is_lower_case=True):\n    tokenizer = ToktokTokenizer()\n    tokens = tokenizer.tokenize(text)\n    tokens = [token.strip() for token in tokens]\n    if is_lower_case:\n        filtered_tokens = [token for token in tokens if token not in stopword_list]\n    else:\n        filtered_tokens = [token for token in tokens if token.lower() not in stopword_list]\n    filtered_text = ' '.join(filtered_tokens)\n    return filtered_text\n\n# remove contractions:\ncontraction_dict = {\"ain't\": \"is not\", \"aren't\": \"are not\",\"can't\": \"cannot\", \"'cause\": \"because\", \"could've\": \"could have\", \"couldn't\": \"could not\", \"didn't\": \"did not\",  \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\", \"he'd\": \"he would\",\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\", \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\",  \"I'd\": \"I would\", \"I'd've\": \"I would have\", \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"I've\": \"I have\", \"i'd\": \"i would\", \"i'd've\": \"i would have\", \"i'll\": \"i will\",  \"i'll've\": \"i will have\",\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'd've\": \"it would have\", \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\", \"ma'am\": \"madam\", \"mayn't\": \"may not\", \"might've\": \"might have\",\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \"must've\": \"must have\", \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \"needn't\": \"need not\", \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\", \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\", \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\", \"she'll\": \"she will\", \"she'll've\": \"she will have\", \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\", \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \"there'd've\": \"there would have\", \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\", \"they'll\": \"they will\", \"they'll've\": \"they will have\", \"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\", \"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\", \"we'll've\": \"we will have\", \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\", \"what're\": \"what are\",  \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\", \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\"you'd\": \"you would\", \"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\"}\n\ndef _get_contractions(contraction_dict):\n    contraction_re = re.compile('(%s)' % '|'.join(contraction_dict.keys()))\n    return contraction_dict, contraction_re\n\ncontractions, contractions_re = _get_contractions(contraction_dict)\n\ndef replace_contractions(text):\n    def replace(match):\n        return contractions[match.group(0)]\n    return contractions_re.sub(replace, text)\n\n# Using lemmatizer to keep dictionary form of words. Might be helpful if later we want to use word embeddings.\nwordnet_lemmatizer = WordNetLemmatizer()\ndef lemma_text(text):\n    tokenizer = ToktokTokenizer()\n    tokens = tokenizer.tokenize(text)\n    tokens = [token.strip() for token in tokens]\n    tokens = [wordnet_lemmatizer.lemmatize(token) for token in tokens]\n    return ' '.join(tokens)\n\n\ndef clean_sentence(x):\n    x = x.lower()\n    x = clean_text(x)\n    x = replace_typical_misspell(x)\n    x = remove_stopwords(x)\n    x = replace_contractions(x)\n    x = lemma_text(x)\n    x = x.replace(\"'\",\"\")\n    return x","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"63cb21525251b060aeb309e7be4b48772f8720f5","scrolled":true,"trusted":true},"cell_type":"code","source":"\ntrain_df = pd.read_csv(\"../input/train.csv\")#[:400000]\ntest_df = pd.read_csv(\"../input/test.csv\")#[:20000]\nprint(\"Train shape : \",train_df.shape)\nprint(\"Test shape : \",test_df.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2b23f5bcfd9bd3d47652ac34454f842ae7f6726a"},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4cd4bd5b2aa754993e4c4fa520f5d62d3f99f034"},"cell_type":"code","source":"# clean the sentences\ntrain_df['cleaned_text'] = train_df['question_text'].apply(lambda x : clean_sentence(x))\ntest_df['cleaned_text'] = test_df['question_text'].apply(lambda x : clean_sentence(x))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5203c177aa11fe6eadccadcd91ae2c8c7d5ef661"},"cell_type":"code","source":"# small function to find threshold and find best f score - Eval metric of competition\ndef bestThresshold(y_train,train_preds):\n    tmp = [0,0,0] # idx, cur, max\n    delta = 0\n    for tmp[0] in tqdm(np.arange(0.1, 0.501, 0.01)):\n        tmp[1] = f1_score(y_train, np.array(train_preds)>tmp[0])\n        if tmp[1] > tmp[2]:\n            delta = tmp[0]\n            tmp[2] = tmp[1]\n    # print('best threshold is {:.4f} with F1 score: {:.4f}'.format(delta, tmp[2]))\n    return tmp[2]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6d0f7bd7abed7262a7c09c2b6bd16665529c60b1"},"cell_type":"markdown","source":"## Lets start with modelling "},{"metadata":{"trusted":true,"_uuid":"bce223df048847b1cdb7d8f04ea7c686350be744"},"cell_type":"code","source":"# HELPER FUNCTIONS\n\ndef model_train_cv(x_train,y_train,nfold,model_obj):\n    splits = list(StratifiedKFold(n_splits=nfold, shuffle=True, random_state=SEED).split(x_train, y_train))\n    x_train = x_train\n    y_train = np.array(y_train)\n    # matrix for the out-of-fold predictions\n    train_oof_preds = np.zeros((x_train.shape[0]))\n    for i, (train_idx, valid_idx) in enumerate(splits):\n\n        x_train_fold = x_train[train_idx.astype(int)]\n        y_train_fold = y_train[train_idx.astype(int)]\n        x_val_fold = x_train[valid_idx.astype(int)]\n        y_val_fold = y_train[valid_idx.astype(int)]\n\n        clf = copy.deepcopy(model_obj)\n        clf.fit(x_train_fold, y_train_fold)\n        valid_preds_fold = clf.predict_proba(x_val_fold)[:,1]\n\n        # storing OOF predictions\n        train_oof_preds[valid_idx] = valid_preds_fold\n    return train_oof_preds\n\ndef lgb_model_train_cv(x_train,y_train,nfold,lgb):\n    splits = list(StratifiedKFold(n_splits=nfold, shuffle=True, random_state=SEED).split(x_train, y_train))\n    x_train = x_train\n    y_train = np.array(y_train)\n    # matrix for the out-of-fold predictions\n    train_oof_preds = np.zeros((x_train.shape[0]))\n    for i, (train_idx, valid_idx) in enumerate(splits):\n        x_train_fold = x_train[train_idx.astype(int)]\n        y_train_fold = y_train[train_idx.astype(int)]\n        x_val_fold = x_train[valid_idx.astype(int)]\n        y_val_fold = y_train[valid_idx.astype(int)]\n        d_train = lgb.Dataset(x_train_fold, label=y_train_fold)\n        d_val = lgb.Dataset(x_val_fold, label=y_val_fold)\n        params = {}\n        params['learning_rate'] = 0.01\n        params['boosting_type'] = 'gbdt'\n        params['objective'] = 'binary'\n        params['metric'] = 'binary_logloss'\n        params['sub_feature'] = 0.5\n        params['num_leaves'] = 10\n        params['min_data'] = 50\n        params['max_depth'] = 10\n        \n        clf = lgb.train(params, d_train, num_boost_round = 100,valid_sets=(d_val), early_stopping_rounds=10,verbose_eval=10)\n        valid_preds_fold = clf.predict(x_val_fold)\n        # storing OOF predictions\n        train_oof_preds[valid_idx] = valid_preds_fold\n    return train_oof_preds","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b785fad987047ecc499fea7557c7e24e0deae872"},"cell_type":"markdown","source":"### 1. Bag of words model using Count Vectorizer:"},{"metadata":{"_uuid":"f717429c36d904902fc406edf7a820943261950e"},"cell_type":"markdown","source":"![count vectorizer](https://mlwhiz.com/images/countvectorizer.png)"},{"metadata":{"trusted":true,"_uuid":"c0889276092d61b1146c2561b52da39448755bdd"},"cell_type":"code","source":"cnt_vectorizer = CountVectorizer(dtype=np.float32,\n            strip_accents='unicode', analyzer='word',token_pattern=r'\\w{1,}',\n            ngram_range=(1, 3),min_df=3)\n\n# Fitting count vectorizer to both training and test sets (semi-supervised learning)\ncnt_vectorizer.fit(list(train_df.cleaned_text.values) + list(test_df.cleaned_text.values))\nxtrain =  cnt_vectorizer.transform(train_df.cleaned_text.values) \n#xtest_cntv = cnt_vectorizer.transform(test_df.cleaned_text.values)\ny_train = train_df.target.values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fbb4b1e370b6ef71a2c79883f5084b62f1d6d31a"},"cell_type":"code","source":"# Fitting a simple Logistic Regression on CountVectorizer Model\ntrain_oof_preds = model_train_cv(xtrain,y_train,5,LogisticRegression(C=1.0))\n\nprint (\"F1 Score: %0.3f \" % bestThresshold(y_train,train_oof_preds))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"620b177a5db5cb39ac36e648fdcadcf941e6758d"},"cell_type":"markdown","source":"We are able to get a F1 local CV score of ___ with our fairly simple model which just counts the number of time some ngrams appear in a sentence. That is pretty good. Let us try Multinomial NB"},{"metadata":{"trusted":true,"_uuid":"ab05c54baa2d43afdd848c346bb1453977755cec"},"cell_type":"code","source":"# fitting a simple Naive Bayes model in place of logistic regression using the same features\ntrain_oof_preds = model_train_cv(xtrain,y_train,5,MultinomialNB())\nprint (\"F1 Score: %0.3f \" % bestThresshold(y_train,train_oof_preds))\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1ff5c6c21fad9b7b849791bdd42d1685d3fe865a"},"cell_type":"markdown","source":"We are able to get a good F1 local CV score  with our fairly simple model which just counts the number of time some ngrams appear in a sentence. \nYou can also try running SVMs which were used extensively when trying out models on Text. But they are pretty slow so not using them here. \n\nLets try LightGBM also."},{"metadata":{"trusted":true,"_uuid":"bad2a873ad97e3c846935859c1b919596b9a5012"},"cell_type":"code","source":"# fitting a simple Naive Bayes model in place of logistic regression using the same features\ntrain_oof_preds = lgb_model_train_cv(xtrain,y_train,5,lgb)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e0b1e974e9750ce032157e3c48144ab46dcaa737"},"cell_type":"code","source":"print (\"F1 Score: %0.3f \" % bestThresshold(y_train,train_oof_preds))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5f5b8c7f8e8148dbc146ae9f5e58cab08cfda8ec","_kg_hide-output":true,"_kg_hide-input":true},"cell_type":"code","source":"xtrain=0\ndel xtrain\n#del xtest_cntv\ndel cnt_vectorizer\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6d9d3e56ef8fedd09b6ce9ce3e86a9aa800ab2ff"},"cell_type":"markdown","source":"### 2. Bag of words model using TFIDF:"},{"metadata":{"_uuid":"d26851473c508638b3c9654e3faaf71eefcab9dc"},"cell_type":"markdown","source":"![tfidf](https://mlwhiz.com/images/tfidf.png)"},{"metadata":{"trusted":true,"_uuid":"2111fcd6ee71a6efe1472dc02cd78748dfe4b131"},"cell_type":"code","source":"# Always start with these features. They work (almost) everytime!\ntfv = TfidfVectorizer(dtype=np.float32, min_df=3,  max_features=None, \n            strip_accents='unicode', analyzer='word',token_pattern=r'\\w{1,}',\n            ngram_range=(1, 3), use_idf=1,smooth_idf=1,sublinear_tf=1,\n            stop_words = 'english')\n\n# Fitting TF-IDF to both training and test sets (semi-supervised learning)\ntfv.fit(list(train_df.cleaned_text.values) + list(test_df.cleaned_text.values))\nxtrain =  tfv.transform(train_df.cleaned_text.values) \n#xtest_tfv = tfv.transform(test_df.cleaned_text.values)\ny_train = train_df.target.values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"02dd3e51009b4abe9e80b4e741281d194151e0f2"},"cell_type":"code","source":"# Fitting a simple Logistic Regression on TFIDF Feats\ntrain_oof_preds = model_train_cv(xtrain,y_train,5,LogisticRegression(C=1.0))\n\nprint (\"F1 Score: %0.3f \" % bestThresshold(y_train,train_oof_preds))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4ba5d035e5c061af0f559d31c6f48da92658498c"},"cell_type":"code","source":"# fitting a simple Naive Bayes model in place of logistic regression using the same features\ntrain_oof_preds = model_train_cv(xtrain,y_train,5,MultinomialNB())\nprint (\"F1 Score: %0.3f \" % bestThresshold(y_train,train_oof_preds))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"05d0264ec38b9d136c86baea691bd551c5d928e4"},"cell_type":"code","source":"# fitting a simple Naive Bayes model in place of logistic regression using the same features\ntrain_oof_preds = lgb_model_train_cv(xtrain,y_train,5,lgb)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"763e18b9517d25142d7d14e1a5af73f439cc6762"},"cell_type":"code","source":"print (\"F1 Score: %0.3f \" % bestThresshold(y_train,train_oof_preds))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"96e41566b94469a09e3f0ce6886fc929607166bc","_kg_hide-output":true,"_kg_hide-input":true},"cell_type":"code","source":"xtrain=0\ndel xtrain\n#del xtest_tfv\ndel tfv\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c0a388d06979b2d326ba5a25b0756e560f8076fc"},"cell_type":"markdown","source":"> ### 3. Bag of words model using Hashing Vectorizer"},{"metadata":{"_uuid":"e26088e95a86774d746a4c8bac44aa8b60ad8749"},"cell_type":"markdown","source":"![](http://)![hashing features](https://mlwhiz.com/images/hashfeats.png)"},{"metadata":{"trusted":true,"_uuid":"dc836bb9bb2813dd444a3cbee60547bf4382e279"},"cell_type":"code","source":"# Always start with these features. They work (almost) everytime!\nhv = HashingVectorizer(dtype=np.float32,\n            strip_accents='unicode', analyzer='word',\n            ngram_range=(1, 3),n_features=2**10,non_negative=True)\n# Fitting Hash Vectorizer to both training and test sets (semi-supervised learning)\nhv.fit(list(train_df.cleaned_text.values) + list(test_df.cleaned_text.values))\nxtrain =  hv.transform(train_df.cleaned_text.values) \n#xtest_hv = hv.transform(test_df.cleaned_text.values)\ny_train = train_df.target.values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e9d818cf4d7f6707c81067a5388f594942ff081d"},"cell_type":"code","source":"# Fitting a simple Logistic Regression on TFIDF Feats\ntrain_oof_preds = model_train_cv(xtrain,y_train,5,LogisticRegression(C=1.0))\n\nprint (\"F1 Score: %0.3f \" % bestThresshold(y_train,train_oof_preds))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2c216d5ea55057a6fe69a0cb2b7f9c804261b3aa"},"cell_type":"code","source":"# fitting a simple Naive Bayes model in place of logistic regression using the same features\ntrain_oof_preds = model_train_cv(xtrain,y_train,5,MultinomialNB())\nprint (\"F1 Score: %0.3f \" % bestThresshold(y_train,train_oof_preds))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b0c9c39c708f0ea20913209dc4201088823d49f4"},"cell_type":"code","source":"# fitting a simple Lgb model in place of logistic regression using the same features\ntrain_oof_preds = lgb_model_train_cv(xtrain,y_train,5,lgb)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aa8fcd7941a85dc6f0d479c68cc8327508c151f7"},"cell_type":"code","source":"print (\"F1 Score: %0.3f \" % bestThresshold(y_train,train_oof_preds))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b08f1609dff27e7ce227e4ec0324b99dcd942066","_kg_hide-output":true,"_kg_hide-input":true},"cell_type":"code","source":"xtrain=0\ndel xtrain\n#del xtest_hv\ndel hv\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"840c3e63b9b7624d5731732f58d1a27efb4c0425"},"cell_type":"markdown","source":"### 4. Word2vec Embeddings "},{"metadata":{"_uuid":"822814b343526fbdf0594ae54d8d897a6ef36364"},"cell_type":"markdown","source":"![word2vec](https://mlwhiz.com/images/word2vec_feats.png)"},{"metadata":{"trusted":true,"_uuid":"bd750906932813ab89ad2825aaa638a4c65dcafd"},"cell_type":"code","source":"# load the GloVe vectors in a dictionary:\ndef load_glove_index():\n    EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')[:300]\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE))\n    return embeddings_index\n\nembeddings_index = load_glove_index()\n\nprint('Found %s word vectors.' % len(embeddings_index))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3e7761598ecc07fee4fb83164215cfc6c47b9573"},"cell_type":"code","source":"from nltk.corpus import stopwords\nstop_words = stopwords.words('english')\ndef sent2vec(s):\n    words = str(s).lower()\n    words = word_tokenize(words)\n    words = [w for w in words if not w in stop_words]\n    words = [w for w in words if w.isalpha()]\n    M = []\n    for w in words:\n        try:\n            M.append(embeddings_index[w])\n        except:\n            continue\n    M = np.array(M)\n    v = M.sum(axis=0)\n    if type(v) != np.ndarray:\n        return np.zeros(300)\n    return v / np.sqrt((v ** 2).sum())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b1e03bd9f951a96a9a698592c6a8616ca43e1e63"},"cell_type":"code","source":"# create sentence vectors using the above function for training and validation set\nxtrain = [sent2vec(x) for x in tqdm(train_df.cleaned_text.values)]\n#xtest_glove = [sent2vec(x) for x in tqdm(test_df.cleaned_text.values)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d613d7ec1938fb36aa7a4fa8991cc97ad142853e","_kg_hide-input":true,"_kg_hide-output":true},"cell_type":"code","source":"embeddings_index = 0\ndel embeddings_index\n# del xtest_glove\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c5aec2a1f71fde41dfabfd3a60e20d5fdb8b8048"},"cell_type":"code","source":"xtrain = np.array(xtrain)\n# xvalid_glove = np.array(xtest_glove)\ny_train = train_df.target.values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"db93436bf02e182efaf4d161d90cfd9a243f1318"},"cell_type":"code","source":"# Fitting a simple Logistic Regression on glove Feats\ntrain_oof_preds = model_train_cv(xtrain,y_train,5,LogisticRegression(C=1.0))\nprint (\"F1 Score: %0.3f \" % bestThresshold(y_train,train_oof_preds))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7f96d15999879f20bd028d6955b9bb3eae711ed9"},"cell_type":"code","source":"# fitting a simple Lgb model in place of logistic regression using the same features\ntrain_oof_preds = lgb_model_train_cv(xtrain,y_train,5,lgb)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b739380b48c7c4bda93f533c1838ddf9d6642c47"},"cell_type":"code","source":"print (\"F1 Score: %0.3f \" % bestThresshold(y_train,train_oof_preds))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f63cda14397adf5dc91a12363c1092a6f3c93543","trusted":false},"cell_type":"markdown","source":"So this is it. All of these models are not tuned yet and could be tuned further to improve performance. But it is good to get sort of baselines and appreciate the sort of performance we can get out of neural nets "},{"metadata":{"_uuid":"10c39195707e4710e11d31bafd4bde1e21ff2d6f","trusted":false},"cell_type":"markdown","source":"References:\n    https://www.kaggle.com/abhishek/approaching-almost-any-nlp-problem-on-kaggle"},{"metadata":{"_uuid":"162772df446ca6ad6dda6a003d8ea2e58106192f","trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fdca46457c79c6b56409e6cfbc76d90de4d19cbe","trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"717f6f2fbaa37d01db018baafcf8216fee98ce6a","trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7a713658e272a3d7ec00267faa5e3d02362269ed","trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7ce9e3159ce70cdaa0d706249cf80ea3157b39d0","trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0fb26d857e36473fe7a458cf61692ceff5cadceb","trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8dee2aefa02222d4dd5e67640708cad59285a796","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8682fc20ab59210e2aba1381232e68ce4e9519d2","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2681fc633782d5a7005a0db95de14c6b05b88dac","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"39e764384d4073282b6a51bcd1ecf382134a33a6","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3703e549d2d8d5bb0948c2771b1e6636f0fccef2","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}