{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-05T12:21:16.521117Z","iopub.execute_input":"2021-06-05T12:21:16.5216Z","iopub.status.idle":"2021-06-05T12:21:16.537026Z","shell.execute_reply.started":"2021-06-05T12:21:16.521492Z","shell.execute_reply":"2021-06-05T12:21:16.535761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/train.csv')","metadata":{"execution":{"iopub.status.busy":"2021-06-05T12:21:52.776285Z","iopub.execute_input":"2021-06-05T12:21:52.776979Z","iopub.status.idle":"2021-06-05T12:21:57.682757Z","shell.execute_reply.started":"2021-06-05T12:21:52.776906Z","shell.execute_reply":"2021-06-05T12:21:57.681458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import nltk\nimport string\nfrom nltk.tokenize import word_tokenize\nfrom nltk.corpus import stopwords\nfrom nltk.stem import WordNetLemmatizer\n\nnltk.download('stopwords')\nnltk_stopwords = stopwords.words('english')\n\nwordnet_lemmatizer = WordNetLemmatizer()\n","metadata":{"execution":{"iopub.status.busy":"2021-06-05T12:22:00.371427Z","iopub.execute_input":"2021-06-05T12:22:00.371941Z","iopub.status.idle":"2021-06-05T12:22:22.099626Z","shell.execute_reply.started":"2021-06-05T12:22:00.371898Z","shell.execute_reply":"2021-06-05T12:22:22.098081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lemSentence(msg):\n    tokens = word_tokenize(msg)\n    lem=[]\n    for i in tokens:\n        lem.append(wordnet_lemmatizer.lemmatize(i,pos=\"v\"))\n        lem.append(\" \")\n    return \"\".join(lem)\ndef clean(msg):\n    msg = msg.translate(str.maketrans('','',string.punctuation))\n    msg = msg.translate(str.maketrans('','',string.digits))\n    msg = [word for word in word_tokenize(msg) if not word.lower() in nltk_stopwords]\n    msg = ' '.join(msg)\n    msg = lemSentence(msg)\n    return msg","metadata":{"execution":{"iopub.status.busy":"2021-06-05T12:22:27.361234Z","iopub.execute_input":"2021-06-05T12:22:27.36161Z","iopub.status.idle":"2021-06-05T12:22:27.369709Z","shell.execute_reply.started":"2021-06-05T12:22:27.361578Z","shell.execute_reply":"2021-06-05T12:22:27.368465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['clean_text'] = df.question_text.apply(lambda x: clean(x))\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-05T12:22:30.736311Z","iopub.execute_input":"2021-06-05T12:22:30.737003Z","iopub.status.idle":"2021-06-05T12:31:09.986172Z","shell.execute_reply.started":"2021-06-05T12:22:30.736946Z","shell.execute_reply":"2021-06-05T12:31:09.985057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nx_train,x_test,y_train,y_test = train_test_split(df['clean_text'],df['target'],test_size=0.2,random_state=42)","metadata":{"execution":{"iopub.status.busy":"2021-06-05T12:43:09.937396Z","iopub.execute_input":"2021-06-05T12:43:09.938076Z","iopub.status.idle":"2021-06-05T12:43:10.710163Z","shell.execute_reply.started":"2021-06-05T12:43:09.938031Z","shell.execute_reply":"2021-06-05T12:43:10.709019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import precision_score, recall_score,f1_score, accuracy_score","metadata":{"execution":{"iopub.status.busy":"2021-06-05T12:43:42.995852Z","iopub.execute_input":"2021-06-05T12:43:42.996273Z","iopub.status.idle":"2021-06-05T12:43:43.001518Z","shell.execute_reply.started":"2021-06-05T12:43:42.996239Z","shell.execute_reply":"2021-06-05T12:43:43.00023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import SGDClassifier\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.naive_bayes import MultinomialNB\nfrom sklearn.pipeline import make_pipeline\nmodel = make_pipeline(TfidfVectorizer(), MultinomialNB(alpha=1))\nmodel.fit(x_train,y_train)\npred = model.predict(x_test)\nprint('f1_score',f1_score(y_test, pred))\nprint('Accuracy: {:.2f}'.format(accuracy_score(y_test, pred)))\nmodel1 = make_pipeline(TfidfVectorizer(), SGDClassifier(random_state=0))\nmodel1.fit(x_train,y_train)\npred1 = model1.predict(x_test)\nprint('f1_score',f1_score(y_test, pred1))\nprint('Accuracy: {:.2f}'.format(accuracy_score(y_test, pred1)))","metadata":{"execution":{"iopub.status.busy":"2021-06-03T13:11:53.429348Z","iopub.execute_input":"2021-06-03T13:11:53.429717Z","iopub.status.idle":"2021-06-03T13:12:31.178025Z","shell.execute_reply.started":"2021-06-03T13:11:53.429681Z","shell.execute_reply":"2021-06-03T13:12:31.176908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import SGDClassifier\nfrom sklearn.naive_bayes import MultinomialNB\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.feature_extraction.text import CountVectorizer\nmodel = make_pipeline(CountVectorizer(), MultinomialNB(alpha=1))\nmodel.fit(x_train,y_train)\npred = model.predict(x_test)\nprint('f1_score',f1_score(y_test, pred))\nprint('Accuracy: {:.2f}'.format(accuracy_score(y_test, pred)))\nmodel1 = make_pipeline(CountVectorizer(), SGDClassifier(random_state=0))\nmodel1.fit(x_train,y_train)\npred1 = model1.predict(x_test)\nprint('f1_score',f1_score(y_test, pred1))\nprint('Accuracy: {:.2f}'.format(accuracy_score(y_test, pred1)))","metadata":{"execution":{"iopub.status.busy":"2021-06-03T13:13:11.249071Z","iopub.execute_input":"2021-06-03T13:13:11.249445Z","iopub.status.idle":"2021-06-03T13:13:48.177756Z","shell.execute_reply.started":"2021-06-03T13:13:11.249416Z","shell.execute_reply":"2021-06-03T13:13:48.176732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.feature_extraction.text import CountVectorizer\n#model3 = make_pipeline(CountVectorizer(), KNeighborsClassifier(n_neighbors=3))\n#model3.fit(x_train, y_train)\n#pred3 = model3.predict(x_test)\n#print('f1_score',f1_score(y_test, pred3))\n#print('Accuracy: {:.2f}'.format(accuracy_score(y_test, pred3)))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-05T12:50:34.076715Z","iopub.execute_input":"2021-06-05T12:50:34.07721Z","iopub.status.idle":"2021-06-05T12:50:34.087336Z","shell.execute_reply.started":"2021-06-05T12:50:34.077166Z","shell.execute_reply":"2021-06-05T12:50:34.085891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import tree\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.pipeline import Pipeline\ncount_vectorizer = CountVectorizer()\nmodel4 = RandomForestClassifier(random_state=0, max_depth=2)\nvectorize_model_pipeline = Pipeline([\n    ('count_vectorizer', count_vectorizer),\n    ('model', model4)])\nvectorize_model_pipeline.fit(x_train, y_train)\n#clf = make_pipeline(CountVectorizer(), tree.DecisionTreeClassifier(random_state=0, max_depth=2))\n#clf.fit(x_train, y_train)\n#tree.plot_tree(clf) \npred4 = vectorize_model_pipeline.predict(x_test)\nprint('f1_score',f1_score(y_test, pred4))\nprint('Accuracy: {:.2f}'.format(accuracy_score(y_test, pred4)))","metadata":{"execution":{"iopub.status.busy":"2021-06-05T13:49:44.995515Z","iopub.execute_input":"2021-06-05T13:49:44.996078Z","iopub.status.idle":"2021-06-05T13:50:18.127547Z","shell.execute_reply.started":"2021-06-05T13:49:44.996027Z","shell.execute_reply":"2021-06-05T13:50:18.126052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nconf_mx = confusion_matrix(y_test, pred1)\nconf_mx","metadata":{"execution":{"iopub.status.busy":"2021-06-03T13:14:09.213634Z","iopub.execute_input":"2021-06-03T13:14:09.214053Z","iopub.status.idle":"2021-06-03T13:14:09.725034Z","shell.execute_reply.started":"2021-06-03T13:14:09.214015Z","shell.execute_reply":"2021-06-03T13:14:09.724146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('recall',recall_score(y_test, pred1))\nprint('precission',precision_score(y_test, pred1))","metadata":{"execution":{"iopub.status.busy":"2021-06-05T13:50:44.700828Z","iopub.execute_input":"2021-06-05T13:50:44.70125Z","iopub.status.idle":"2021-06-05T13:50:44.863758Z","shell.execute_reply.started":"2021-06-05T13:50:44.701213Z","shell.execute_reply":"2021-06-05T13:50:44.862422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tn, fp, fn, tp = confusion_matrix(y_test, pred1).ravel()\nprint(\"True Negatives: \",tn)\nprint(\"False Positives: \",fp)\nprint(\"False Negatives: \",fn)\nprint(\"True Positives: \",tp)\n","metadata":{"execution":{"iopub.status.busy":"2021-06-02T18:58:20.418376Z","iopub.execute_input":"2021-06-02T18:58:20.418799Z","iopub.status.idle":"2021-06-02T18:58:20.945389Z","shell.execute_reply.started":"2021-06-02T18:58:20.418764Z","shell.execute_reply":"2021-06-02T18:58:20.944286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"precission = tp/(tp+fp)\nrecall=tp/(tp+fn)\nprint(recall)\nprint(precission)\n","metadata":{"execution":{"iopub.status.busy":"2021-06-02T18:58:23.96182Z","iopub.execute_input":"2021-06-02T18:58:23.962189Z","iopub.status.idle":"2021-06-02T18:58:23.9679Z","shell.execute_reply.started":"2021-06-02T18:58:23.962152Z","shell.execute_reply":"2021-06-02T18:58:23.966945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tp, fn, fp, tn = confusion_matrix(y_test, pred1).ravel()\nprint(\"True Negatives: \",tn)\nprint(\"False Positives: \",fp)\nprint(\"False Negatives: \",fn)\nprint(\"True Positives: \",tp)","metadata":{"execution":{"iopub.status.busy":"2021-06-02T18:58:36.349778Z","iopub.execute_input":"2021-06-02T18:58:36.350193Z","iopub.status.idle":"2021-06-02T18:58:36.865779Z","shell.execute_reply.started":"2021-06-02T18:58:36.350144Z","shell.execute_reply":"2021-06-02T18:58:36.864701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"precission = tp/(tp+fp)\nrecall=tp/(tp+fn)\nprint(recall)\nprint(precission)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport itertools    \ndef plot_confusion_matrix(cm, classes,\n                          normalize=False,\n                          title='Confusion matrix'\n                          ):\n    \"\"\"\n    This function prints and plots the confusion matrix.\n    Normalization can be applied by setting `normalize=True`.\n    \"\"\"\n    if normalize:\n        cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n        print(\"Normalized confusion matrix\")\n    else:\n        print('Confusion matrix, without normalization')\n\n    print(cm)\n\n    plt.imshow(cm, interpolation='nearest', vmin = 0, \n                             vmax = 40000)\n    plt.title(title)\n    plt.colorbar()\n    tick_marks = np.arange(len(classes))\n    plt.xticks(tick_marks, classes, rotation=45)\n    plt.yticks(tick_marks, classes)\n\n    fmt = '.2f' if normalize else 'd'\n    thresh = cm.max() / 2.\n    for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])):\n        plt.text(j, i, format(cm[i, j], fmt),\n                 horizontalalignment=\"center\",\n                 color=\"black\")\n\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')\n    plt.tight_layout()\n\n\n# Compute confusion matrix\ncnf_matrix = confusion_matrix(y_test, pred1)\nnp.set_printoptions(precision=2)\n\n# Plot non-normalized confusion matrix\nplt.figure()\nplot_confusion_matrix(cnf_matrix, classes=['0','1'],\n                      title='Confusion matrix, without normalization')","metadata":{"execution":{"iopub.status.busy":"2021-06-03T13:16:14.249955Z","iopub.execute_input":"2021-06-03T13:16:14.25036Z","iopub.status.idle":"2021-06-03T13:16:15.066076Z","shell.execute_reply.started":"2021-06-03T13:16:14.250329Z","shell.execute_reply":"2021-06-03T13:16:15.065107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conf_mx1 = confusion_matrix(y_test, pred, normalize='true')\nprint(conf_mx1)\ntn, fp, fn, tp = conf_mx1.ravel()\nprint(\"True Negatives: \",tn)\nprint(\"False Positives: \",fp)\nprint(\"False Negatives: \",fn)\nprint(\"True Positives: \",tp)\n","metadata":{"execution":{"iopub.status.busy":"2021-06-01T12:35:57.741256Z","iopub.execute_input":"2021-06-01T12:35:57.741528Z","iopub.status.idle":"2021-06-01T12:35:58.030952Z","shell.execute_reply.started":"2021-06-01T12:35:57.741507Z","shell.execute_reply":"2021-06-01T12:35:58.029797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\ndef plot_confusion_matrix(matrix):\n    \"\"\"If you prefer color and a colorbar\"\"\"\n    fig = plt.figure(figsize=(8,8))\n    ax = fig.add_subplot(111)\n    cax = ax.matshow(matrix, interpolation='nearest')\n    plt.imshow(matrix, vmin = 0, \n                             vmax = 40000)\n    fig.colorbar(cax)\n    plt.ylabel(\"True label\")\n    plt.xlabel(\"Predicted label\")","metadata":{"execution":{"iopub.status.busy":"2021-06-01T12:56:19.491735Z","iopub.execute_input":"2021-06-01T12:56:19.492049Z","iopub.status.idle":"2021-06-01T12:56:19.498779Z","shell.execute_reply.started":"2021-06-01T12:56:19.492025Z","shell.execute_reply":"2021-06-01T12:56:19.497402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_confusion_matrix(conf_mx)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_confusion_matrix(conf_mx1)","metadata":{"execution":{"iopub.status.busy":"2021-06-01T12:36:13.181934Z","iopub.execute_input":"2021-06-01T12:36:13.182245Z","iopub.status.idle":"2021-06-01T12:36:13.588656Z","shell.execute_reply.started":"2021-06-01T12:36:13.18221Z","shell.execute_reply":"2021-06-01T12:36:13.587603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.matshow(conf_mx)\nplt.ylabel(\"True label\")\nplt.xlabel(\"Predicted label\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-01T12:20:12.416754Z","iopub.execute_input":"2021-06-01T12:20:12.417029Z","iopub.status.idle":"2021-06-01T12:20:12.547552Z","shell.execute_reply.started":"2021-06-01T12:20:12.417006Z","shell.execute_reply":"2021-06-01T12:20:12.546567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import precision_score, recall_score,f1_score\nprecision_score(y_test, pred)","metadata":{"execution":{"iopub.status.busy":"2021-06-01T12:04:40.230788Z","iopub.execute_input":"2021-06-01T12:04:40.231368Z","iopub.status.idle":"2021-06-01T12:04:40.353193Z","shell.execute_reply.started":"2021-06-01T12:04:40.231335Z","shell.execute_reply":"2021-06-01T12:04:40.352169Z"},"trusted":true},"execution_count":null,"outputs":[]}]}