{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Table of Content:\n\n1. Load the Data\n2. Quick view of Data\n3. EDA\n    * Data Visualization\n    * Text Preprocessing\n    * Word Embeddings\n4. Build Basic Linear Model\n5. Evaluate the results\n","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport time\nfrom tqdm import tqdm\nimport itertools\nimport h2o\n\nimport lightgbm as lgb\n\nfrom sklearn.metrics import f1_score\nfrom sklearn.model_selection import KFold\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.naive_bayes import GaussianNB, MultinomialNB, BernoulliNB\nfrom sklearn.decomposition import PCA, TruncatedSVD\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.feature_extraction.text import CountVectorizer, TfidfVectorizer\nfrom sklearn.metrics import confusion_matrix\n\nimport nltk\nfrom nltk.corpus import stopwords\nimport string\nimport re\nfrom nltk.util import ngrams\nfrom collections import Counter\nfrom collections import defaultdict\nfrom spacy.lang.en.stop_words import STOP_WORDS\n\nfrom scipy.sparse import hstack\n\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport matplotlib.patches as mpatches\n\nfrom IPython.core.display import display, HTML\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nplt.style.use('fivethirtyeight')\n\n%matplotlib inline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-06T10:46:54.533111Z","iopub.execute_input":"2022-03-06T10:46:54.534314Z","iopub.status.idle":"2022-03-06T10:46:59.940197Z","shell.execute_reply.started":"2022-03-06T10:46:54.534153Z","shell.execute_reply":"2022-03-06T10:46:59.939474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2022-03-06T10:46:59.941610Z","iopub.execute_input":"2022-03-06T10:46:59.942633Z","iopub.status.idle":"2022-03-06T10:46:59.950435Z","shell.execute_reply.started":"2022-03-06T10:46:59.942578Z","shell.execute_reply":"2022-03-06T10:46:59.949369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1. Load the Data","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/quora-insincere-questions-classification//train.csv').fillna(' ')\ntest = pd.read_csv('/kaggle/input/quora-insincere-questions-classification//test.csv').fillna(' ')","metadata":{"execution":{"iopub.status.busy":"2022-03-06T10:58:21.974328Z","iopub.execute_input":"2022-03-06T10:58:21.974686Z","iopub.status.idle":"2022-03-06T10:58:26.488116Z","shell.execute_reply.started":"2022-03-06T10:58:21.974654Z","shell.execute_reply":"2022-03-06T10:58:26.486548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_df(dfs:list, captions:list):\n    \"\"\"Display tables side by side to save vertical space\n    Input:\n        dfs: list of pandas.DataFrame\n        captions: list of table captions\n    \"\"\"\n    output = \"\"\n    combined = dict(zip(captions, dfs))\n    for caption, df in combined.items():\n        output += df.style.set_table_attributes(\"style='display:inline'\").set_caption(caption)._repr_html_()\n        output += \"\\xa0\\xa0\\xa0\"\n    display(HTML(output))","metadata":{"execution":{"iopub.status.busy":"2022-03-06T10:58:26.490098Z","iopub.execute_input":"2022-03-06T10:58:26.490400Z","iopub.status.idle":"2022-03-06T10:58:26.499216Z","shell.execute_reply.started":"2022-03-06T10:58:26.490362Z","shell.execute_reply":"2022-03-06T10:58:26.497781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Quick view of Data","metadata":{}},{"cell_type":"code","source":"display_df([train.sample(5), test.sample(5)], ['Train', 'Test'])","metadata":{"execution":{"iopub.status.busy":"2022-03-06T10:58:26.502251Z","iopub.execute_input":"2022-03-06T10:58:26.502640Z","iopub.status.idle":"2022-03-06T10:58:26.793806Z","shell.execute_reply.started":"2022-03-06T10:58:26.502590Z","shell.execute_reply":"2022-03-06T10:58:26.792735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Train shape: {train.shape} ||  Test shape:{test.shape}\")","metadata":{"execution":{"iopub.status.busy":"2022-03-06T10:47:05.353469Z","iopub.execute_input":"2022-03-06T10:47:05.353713Z","iopub.status.idle":"2022-03-06T10:47:05.359801Z","shell.execute_reply.started":"2022-03-06T10:47:05.353681Z","shell.execute_reply":"2022-03-06T10:47:05.358353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. EDA\n **Data Visualization**","metadata":{}},{"cell_type":"code","source":"train['target'].value_counts(normalize=True)*100","metadata":{"execution":{"iopub.status.busy":"2022-03-06T10:47:05.361789Z","iopub.execute_input":"2022-03-06T10:47:05.362451Z","iopub.status.idle":"2022-03-06T10:47:05.387839Z","shell.execute_reply.started":"2022-03-06T10:47:05.362374Z","shell.execute_reply":"2022-03-06T10:47:05.387277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,5))\nsns.countplot(data=train, x='target');","metadata":{"execution":{"iopub.status.busy":"2022-03-06T10:47:05.389010Z","iopub.execute_input":"2022-03-06T10:47:05.389525Z","iopub.status.idle":"2022-03-06T10:47:05.731847Z","shell.execute_reply.started":"2022-03-06T10:47:05.389487Z","shell.execute_reply":"2022-03-06T10:47:05.730931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(1,len(train.target.unique()), figsize=(20,8))\nfig.suptitle('Unigram Analysis')\n\nfor index,target in enumerate(train.target.unique()):\n    dct=defaultdict(int) \n    curdf=train[train['target']==target]  \n    allwordsarr=curdf.question_text.str.cat().split()\n    counter=Counter(allwordsarr)\n    most=counter.most_common()\n    x=[]\n    y=[]\n    for word,count in most[:100]:\n        if (word.lower() not in STOP_WORDS):\n            x.append(word)\n            y.append(count)\n    sns.barplot(ax=axes[index],x=y,y=x)\n    axes[index].set_title(\"Target: \"+str(target))","metadata":{"execution":{"iopub.status.busy":"2022-03-06T10:47:05.733371Z","iopub.execute_input":"2022-03-06T10:47:05.733615Z","iopub.status.idle":"2022-03-06T10:47:13.641918Z","shell.execute_reply.started":"2022-03-06T10:47:05.733584Z","shell.execute_reply":"2022-03-06T10:47:13.640570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_top_bigrams(corpus, n=None):\n    vec = CountVectorizer(ngram_range=(2, 2)).fit(corpus)\n    bag_of_words = vec.transform(corpus)\n    sum_words = bag_of_words.sum(axis=0) \n    words_freq = [(word, sum_words[0, idx]) for word, idx in vec.vocabulary_.items()]\n    words_freq =sorted(words_freq, key = lambda x: x[1], reverse=True)\n    return words_freq[:n]\n\nprint(\"Bigram analysis\")\n\nfig, axes = plt.subplots(1,len(train.target.unique()), figsize=(20,8))\nfig.suptitle('Bigram analysis')\n\nfor index,target in enumerate(train.target.unique()):\n    dct=defaultdict(int) \n    top_bigrams=get_top_bigrams(train[train['target']==target].question_text)[:50]\n    x,y=map(list,zip(*top_bigrams))\n    sns.barplot(ax=axes[index],x=y,y=x)\n    axes[index].set_title(\"Target: \"+str(target))\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-06T10:47:13.643886Z","iopub.execute_input":"2022-03-06T10:47:13.644293Z","iopub.status.idle":"2022-03-06T10:48:56.205138Z","shell.execute_reply.started":"2022-03-06T10:47:13.644220Z","shell.execute_reply":"2022-03-06T10:48:56.203857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. EDA\n **Text Preprocessing**","metadata":{}},{"cell_type":"code","source":"#Remove bad symbols and stopwords from test and train data\nREPLACE_BY_SPACE_RE = re.compile('[/(){}\\[\\]\\|@,;]')\nBAD_SYMBOLS_RE = re.compile('[^0-9a-z #+_]')\nSTOPWORDS = set(stopwords.words('english'))\n\ndef text_prepare(text):\n    \"\"\"\n        text: a string\n        \n        return: modified initial string\n    \"\"\"\n    text = text.lower()   # lowercase text\n    text = REPLACE_BY_SPACE_RE.sub(\" \", text)     # replace REPLACE_BY_SPACE_RE symbols by space in text\n    text = BAD_SYMBOLS_RE.sub(\"\", text)     # delete symbols which are in BAD_SYMBOLS_RE from text\n    \n    \n    resultwords = [word for word in text.split() if word not in STOPWORDS]  # delete stopwords from text\n    text = ' '.join(resultwords)\n    \n    return text","metadata":{"execution":{"iopub.status.busy":"2022-03-06T10:48:56.207016Z","iopub.execute_input":"2022-03-06T10:48:56.207254Z","iopub.status.idle":"2022-03-06T10:48:56.221901Z","shell.execute_reply.started":"2022-03-06T10:48:56.207226Z","shell.execute_reply":"2022-03-06T10:48:56.221196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['question_text_cleaned'] = [text_prepare(x) for x in train['question_text']]","metadata":{"execution":{"iopub.status.busy":"2022-03-06T10:48:56.223082Z","iopub.execute_input":"2022-03-06T10:48:56.223435Z","iopub.status.idle":"2022-03-06T10:49:06.141719Z","shell.execute_reply.started":"2022-03-06T10:48:56.223399Z","shell.execute_reply":"2022-03-06T10:49:06.140601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_df([pd.DataFrame(train['question_text']).sample(5, random_state=42), \n            pd.DataFrame(train['question_text_cleaned']).sample(5, random_state=42)], ['Raw Data', 'Cleaned Data'])","metadata":{"execution":{"iopub.status.busy":"2022-03-06T10:49:06.144754Z","iopub.execute_input":"2022-03-06T10:49:06.145007Z","iopub.status.idle":"2022-03-06T10:49:06.313752Z","shell.execute_reply.started":"2022-03-06T10:49:06.144976Z","shell.execute_reply":"2022-03-06T10:49:06.312896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. EDA\n **Word Embeddings**\n> CountVectorizer  ","metadata":{}},{"cell_type":"code","source":"def cv(data):\n    count_vectorizer = CountVectorizer()\n    emb = count_vectorizer.fit_transform(data)\n    return emb, count_vectorizer","metadata":{"execution":{"iopub.status.busy":"2022-03-06T10:49:06.314992Z","iopub.execute_input":"2022-03-06T10:49:06.315619Z","iopub.status.idle":"2022-03-06T10:49:06.321408Z","shell.execute_reply.started":"2022-03-06T10:49:06.315582Z","shell.execute_reply":"2022-03-06T10:49:06.320204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_corpus = train['question_text_cleaned'].tolist()\nlist_labels = train['target'].tolist()\n\nX_train, X_test, y_train, y_test = train_test_split(list_corpus, list_labels, test_size=0.2, \n                                                                                random_state=40)","metadata":{"execution":{"iopub.status.busy":"2022-03-06T10:49:06.323283Z","iopub.execute_input":"2022-03-06T10:49:06.323655Z","iopub.status.idle":"2022-03-06T10:49:07.338313Z","shell.execute_reply.started":"2022-03-06T10:49:06.323613Z","shell.execute_reply":"2022-03-06T10:49:07.337241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_counts, count_vectorizer = cv(X_train)\nX_test_counts = count_vectorizer.transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-03-06T10:49:07.339624Z","iopub.execute_input":"2022-03-06T10:49:07.339891Z","iopub.status.idle":"2022-03-06T10:49:24.013368Z","shell.execute_reply.started":"2022-03-06T10:49:07.339859Z","shell.execute_reply":"2022-03-06T10:49:24.012521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list(count_vectorizer.vocabulary_.items())[:10]","metadata":{"execution":{"iopub.status.busy":"2022-03-06T10:49:24.015200Z","iopub.execute_input":"2022-03-06T10:49:24.015811Z","iopub.status.idle":"2022-03-06T10:49:24.070519Z","shell.execute_reply.started":"2022-03-06T10:49:24.015705Z","shell.execute_reply":"2022-03-06T10:49:24.069305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Build Basic Linear Model","metadata":{}},{"cell_type":"code","source":"def plot_LSA(test_data, test_labels, savepath=\"PCA.csv\", plot=True):\n        lsa = TruncatedSVD(n_components=2)\n        lsa.fit(test_data)\n        lsa_scores = lsa.transform(test_data)\n        color_mapper = {label:idx for idx,label in enumerate(set(test_labels))}\n        color_column = [color_mapper[label] for label in test_labels]\n        colors = ['blue','red']\n        if plot:\n            plt.scatter(lsa_scores[:,0], lsa_scores[:,1], s=8, alpha=.8, c=test_labels, \n                        cmap=matplotlib.colors.ListedColormap(colors))\n            red_patch = mpatches.Patch(color='blue', label='0')\n            green_patch = mpatches.Patch(color='red', label='1')\n            plt.legend(handles=[red_patch, green_patch], prop={'size': 25})","metadata":{"execution":{"iopub.status.busy":"2022-03-06T10:49:24.072690Z","iopub.execute_input":"2022-03-06T10:49:24.073050Z","iopub.status.idle":"2022-03-06T10:49:24.085861Z","shell.execute_reply.started":"2022-03-06T10:49:24.073006Z","shell.execute_reply":"2022-03-06T10:49:24.084891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(10, 10))          \nplot_LSA(X_train_counts, y_train)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-06T10:49:24.086969Z","iopub.execute_input":"2022-03-06T10:49:24.087638Z","iopub.status.idle":"2022-03-06T10:49:54.787314Z","shell.execute_reply.started":"2022-03-06T10:49:24.087604Z","shell.execute_reply":"2022-03-06T10:49:54.786343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf = LogisticRegression(C=0.5, class_weight='balanced', solver='sag', n_jobs=-1, random_state=40)\nclf.fit(X_train_counts, y_train)\n\ny_predicted_counts = clf.predict(X_test_counts)","metadata":{"execution":{"iopub.status.busy":"2022-03-06T10:49:54.788868Z","iopub.execute_input":"2022-03-06T10:49:54.789232Z","iopub.status.idle":"2022-03-06T10:51:07.974765Z","shell.execute_reply.started":"2022-03-06T10:49:54.789199Z","shell.execute_reply":"2022-03-06T10:51:07.973512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def opt_f1(y_test, y_pred):\n    opt_prob = None\n    f1_max = 0\n\n    for thresh in np.arange(0.1, 0.501, 0.01):\n        thresh = np.round(thresh, 2)\n        f1 = f1_score(y_test, (y_pred > thresh).astype(int))\n        print('F1 score at threshold {} is {}'.format(thresh, f1))\n\n        if f1 > f1_max:\n            f1_max = f1\n            opt_prob = thresh\n\n    print('Optimal probabilty threshold is {} for maximum F1 score {}'.format(opt_prob, f1_max))\n    return opt_prob","metadata":{"execution":{"iopub.status.busy":"2022-03-06T10:51:07.976801Z","iopub.execute_input":"2022-03-06T10:51:07.977164Z","iopub.status.idle":"2022-03-06T10:51:07.985761Z","shell.execute_reply.started":"2022-03-06T10:51:07.977118Z","shell.execute_reply":"2022-03-06T10:51:07.984379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"F1 Score: \", f1_score(y_test, y_predicted_counts))","metadata":{"execution":{"iopub.status.busy":"2022-03-06T10:51:07.987605Z","iopub.execute_input":"2022-03-06T10:51:07.987887Z","iopub.status.idle":"2022-03-06T10:51:08.286287Z","shell.execute_reply.started":"2022-03-06T10:51:07.987855Z","shell.execute_reply":"2022-03-06T10:51:08.285016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_confusion_matrix(cm, classes,\n                          normalize=False,\n                          title='Confusion matrix',\n                          cmap=plt.cm.viridis):\n    if normalize:\n        cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n    plt.imshow(cm, interpolation='nearest', cmap=cmap)\n    plt.title(title, fontsize=20)\n    plt.colorbar()\n    tick_marks = np.arange(len(classes))\n    plt.xticks(tick_marks, classes, fontsize=15)\n    plt.yticks(tick_marks, classes, fontsize=15)\n    \n    fmt = '.2f' if normalize else 'd'\n    thresh = cm.max() / 2.\n\n    for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])):\n        plt.text(j, i, format(cm[i, j], fmt), horizontalalignment=\"center\", \n                 color=\"white\" if cm[i, j] < thresh else \"black\", fontsize=30)\n    \n    plt.tight_layout()\n    plt.ylabel('True label', fontsize=20)\n    plt.xlabel('Predicted label', fontsize=20)\n\n    return plt","metadata":{"execution":{"iopub.status.busy":"2022-03-06T10:51:08.287846Z","iopub.execute_input":"2022-03-06T10:51:08.288111Z","iopub.status.idle":"2022-03-06T10:51:08.299129Z","shell.execute_reply.started":"2022-03-06T10:51:08.288080Z","shell.execute_reply":"2022-03-06T10:51:08.298167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. Evaluate the results","metadata":{}},{"cell_type":"code","source":"cm = confusion_matrix(y_test, y_predicted_counts)\nfig = plt.figure(figsize=(7, 7))\nplot = plot_confusion_matrix(cm, classes=['OK','Toxic'], normalize=False, title='Confusion matrix')\nplt.show();","metadata":{"execution":{"iopub.status.busy":"2022-03-06T10:51:08.300619Z","iopub.execute_input":"2022-03-06T10:51:08.300887Z","iopub.status.idle":"2022-03-06T10:51:08.969073Z","shell.execute_reply.started":"2022-03-06T10:51:08.300855Z","shell.execute_reply":"2022-03-06T10:51:08.967879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}