{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Title: Quora-insincere-questions-classification**","metadata":{}},{"cell_type":"markdown","source":"# **Presentors : Elad Shaked and Omer Shlomo**","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is definimport pandas as pd\n# For example, here's several helpful packages to load\n\nfrom transformers import BertTokenizer, BertForSequenceClassification\nimport numpy as np\nimport pylab as pl\nimport pandas as pd\nimport matplotlib.pyplot as plt \n%matplotlib inline\nimport seaborn as sns\nfrom sklearn.utils import shuffle\nfrom sklearn.svm import SVC\nfrom sklearn.metrics import confusion_matrix,classification_report\nfrom sklearn.model_selection import cross_val_score, GridSearchCV\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\n\nfrom sklearn.metrics import accuracy_score\nfrom keras.models import Sequential\nfrom keras.layers import Dense , Embedding , SimpleRNN\nfrom keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\n\nfrom plotly import tools\nimport plotly.offline as py\npy.init_notebook_mode(connected=True)\nimport plotly.graph_objs as go\n\nimport os\nimport json\nimport string\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport nltk\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize\nfrom nltk.probability import FreqDist\nfrom nltk.util import ngrams\nfrom nltk.stem import WordNetLemmatizer\nfrom wordcloud import WordCloud\nimport re\nfrom sklearn import datasets, svm, metrics\n# Download the required resources\nnltk.download('punkt')\nnltk.download('stopwords')\nnltk.download('wordnet')\n\nimport random\nimport re\nimport string\nimport spacy\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize\n\nimport torch\nimport transformers as ppb\nimport warnings\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-12T21:39:06.485882Z","iopub.execute_input":"2023-06-12T21:39:06.486927Z","iopub.status.idle":"2023-06-12T21:39:24.886040Z","shell.execute_reply.started":"2023-06-12T21:39:06.486889Z","shell.execute_reply":"2023-06-12T21:39:24.884834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n\nimport tensorflow_hub as hub\nimport tensorflow_datasets as tfds\n!pip install transformers datasets evaluate\n","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:39:24.888311Z","iopub.execute_input":"2023-06-12T21:39:24.888789Z","iopub.status.idle":"2023-06-12T21:39:42.232946Z","shell.execute_reply.started":"2023-06-12T21:39:24.888725Z","shell.execute_reply":"2023-06-12T21:39:42.231549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Loading The Data**","metadata":{}},{"cell_type":"code","source":"\n# This takes a few minutes to run, so go grab a tea or coffee while you wait :)\ntrain = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/train.csv')\ntest = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/test.csv')\nsubmission= pd.read_csv('/kaggle/input/quora-insincere-questions-classification/sample_submission.csv')\n","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:39:42.235012Z","iopub.execute_input":"2023-06-12T21:39:42.235424Z","iopub.status.idle":"2023-06-12T21:39:48.803683Z","shell.execute_reply.started":"2023-06-12T21:39:42.235390Z","shell.execute_reply":"2023-06-12T21:39:48.802765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:39:48.805986Z","iopub.execute_input":"2023-06-12T21:39:48.806800Z","iopub.status.idle":"2023-06-12T21:39:48.841640Z","shell.execute_reply.started":"2023-06-12T21:39:48.806765Z","shell.execute_reply":"2023-06-12T21:39:48.840444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:39:48.843009Z","iopub.execute_input":"2023-06-12T21:39:48.843345Z","iopub.status.idle":"2023-06-12T21:39:49.704794Z","shell.execute_reply.started":"2023-06-12T21:39:48.843318Z","shell.execute_reply":"2023-06-12T21:39:49.703313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import psutil\n\n# Process.memory_info is expressed in bytes, so convert to megabytes\nprint(f\"RAM used: {psutil.Process().memory_info().rss / (1024 * 1024):.2f} MB\")","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:39:49.706382Z","iopub.execute_input":"2023-06-12T21:39:49.706784Z","iopub.status.idle":"2023-06-12T21:39:49.714054Z","shell.execute_reply.started":"2023-06-12T21:39:49.706736Z","shell.execute_reply":"2023-06-12T21:39:49.712836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Preproccesing**","metadata":{}},{"cell_type":"code","source":"def preprocess_text(text):\n    # Convert text to lowercase\n    text = text.lower()\n    \n    # Tokenize text into individual words\n    words = word_tokenize(text)\n    \n    # Remove stopwords\n    words = [word for word in words if word not in stopwords.words('english')]\n    \n    # Remove punctuation\n    words = [word for word in words if word not in string.punctuation]\n    \n    # Join the words back into a single string\n    preprocessed_text = ' '.join(words)\n    \n    return preprocessed_text","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:39:49.715649Z","iopub.execute_input":"2023-06-12T21:39:49.716122Z","iopub.status.idle":"2023-06-12T21:39:49.727214Z","shell.execute_reply.started":"2023-06-12T21:39:49.716083Z","shell.execute_reply":"2023-06-12T21:39:49.726077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.target.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:39:49.728935Z","iopub.execute_input":"2023-06-12T21:39:49.729364Z","iopub.status.idle":"2023-06-12T21:39:49.758514Z","shell.execute_reply.started":"2023-06-12T21:39:49.729326Z","shell.execute_reply":"2023-06-12T21:39:49.757282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=train[:10000]\ndf['question_text'] = df['question_text'].apply(lambda d : preprocess_text(d))","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:39:49.760210Z","iopub.execute_input":"2023-06-12T21:39:49.760631Z","iopub.status.idle":"2023-06-12T21:40:12.674080Z","shell.execute_reply.started":"2023-06-12T21:39:49.760591Z","shell.execute_reply":"2023-06-12T21:40:12.672780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cnt_srs = df['target'].value_counts()\ntrace = go.Bar(\n    x=cnt_srs.index,\n    y=cnt_srs.values,\n    marker=dict(\n        color=cnt_srs.values,\n        colorscale = 'Picnic',\n        reversescale = True\n    ),\n)\n\nlayout = go.Layout(\n    title='Target Count',\n    font=dict(size=18)\n)\n\ndata = [trace]\nfig = go.Figure(data=data, layout=layout)\npy.iplot(fig, filename=\"TargetCount\")\n\n## target distribution ##\nlabels = (np.array(cnt_srs.index))\nsizes = (np.array((cnt_srs / cnt_srs.sum())*100))\n\ntrace = go.Pie(labels=labels, values=sizes)\nlayout = go.Layout(\n    title='Target distribution',\n    font=dict(size=18),\n    width=600,\n    height=600,\n)\ndata = [trace]\nfig = go.Figure(data=data, layout=layout)\npy.iplot(fig, filename=\"usertype\")","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:40:12.679988Z","iopub.execute_input":"2023-06-12T21:40:12.680375Z","iopub.status.idle":"2023-06-12T21:40:14.956098Z","shell.execute_reply.started":"2023-06-12T21:40:12.680346Z","shell.execute_reply":"2023-06-12T21:40:14.954976Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport json\nimport string\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize\nfrom nltk.probability import FreqDist\nfrom nltk.util import ngrams\nfrom wordcloud import WordCloud","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:40:14.957786Z","iopub.execute_input":"2023-06-12T21:40:14.958486Z","iopub.status.idle":"2023-06-12T21:40:14.965810Z","shell.execute_reply.started":"2023-06-12T21:40:14.958445Z","shell.execute_reply":"2023-06-12T21:40:14.964714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **N grams**","metadata":{}},{"cell_type":"code","source":"\n\npd.options.mode.chained_assignment = None\npd.options.display.max_columns = 999\n\nstop_words = set(stopwords.words('english'))\n\nprint(\"Train shape:\", train.shape)\nprint(\"Test shape:\", test.shape)\n\n\ndef clean_text(text):\n    text = text.lower()\n    text = re.sub(r\"[^a-zA-Z]\", \" \", text)\n    text = word_tokenize(text)\n    text = [word for word in text if word not in stop_words]\n    text = ' '.join(text)\n    return text\n\ntrain['cleaned_text'] = train['question_text'].apply(clean_text)\n\n\ndef generate_ngrams(text, n):\n    tokens = word_tokenize(text)\n    ngrams_list = list(ngrams(tokens, n))\n    return [' '.join(gram) for gram in ngrams_list]\n\ndef plot_ngrams_freq_dist(text, n, title=None):\n    ngrams_list = generate_ngrams(text, n)\n    fdist = FreqDist(ngrams_list)\n    \n    plt.figure(figsize=(12, 6))\n#     fdist.plot(30)\n    plt.title(title, fontsize=18)\n    plt.show()\nplot_ngrams_freq_dist(' '.join(train[train['target'] == 1]['cleaned_text']), 3, 'Frequency Distribution of Bigrams')\nplot_ngrams_freq_dist(' '.join(train[train['target'] == 1]['cleaned_text']), 2, 'Frequency Distribution of Bigrams')\nplot_ngrams_freq_dist(' '.join(train[train['target'] == 1]['cleaned_text']), 1, 'Frequency Distribution of Bigrams')","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:40:14.967497Z","iopub.execute_input":"2023-06-12T21:40:14.970019Z","iopub.status.idle":"2023-06-12T21:45:23.587923Z","shell.execute_reply.started":"2023-06-12T21:40:14.969986Z","shell.execute_reply":"2023-06-12T21:45:23.587026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **X = text y= target**","metadata":{}},{"cell_type":"code","source":"X = df['question_text']\ny = df['target'][:10000].values\n\nmax_words = 10000\ntokenizer =  Tokenizer(num_words=max_words)\ntokenizer.fit_on_texts(X)\nsequences = tokenizer.texts_to_sequences(X)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:45:23.589189Z","iopub.execute_input":"2023-06-12T21:45:23.589863Z","iopub.status.idle":"2023-06-12T21:45:23.951403Z","shell.execute_reply.started":"2023-06-12T21:45:23.589829Z","shell.execute_reply":"2023-06-12T21:45:23.950226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_sequence_length = max(len(seq) for seq in sequences )\nX = pad_sequences(sequences , maxlen=max_sequence_length )\n","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:45:23.952907Z","iopub.execute_input":"2023-06-12T21:45:23.953217Z","iopub.status.idle":"2023-06-12T21:45:24.009001Z","shell.execute_reply.started":"2023-06-12T21:45:23.953191Z","shell.execute_reply":"2023-06-12T21:45:24.007814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **RNN**","metadata":{}},{"cell_type":"code","source":"x_train , x_test , y_train , y_test =train_test_split(X, y , shuffle=True , test_size=0.2 )\n","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:45:24.010423Z","iopub.execute_input":"2023-06-12T21:45:24.010806Z","iopub.status.idle":"2023-06-12T21:45:24.017481Z","shell.execute_reply.started":"2023-06-12T21:45:24.010766Z","shell.execute_reply":"2023-06-12T21:45:24.016461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model =  Sequential()\nmodel.add(Embedding(max_words , 32 , input_length= max_sequence_length))\nmodel.add(SimpleRNN(32))\nmodel.add(Dense(1 , activation='sigmoid'))","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:45:24.019334Z","iopub.execute_input":"2023-06-12T21:45:24.019847Z","iopub.status.idle":"2023-06-12T21:45:24.330329Z","shell.execute_reply.started":"2023-06-12T21:45:24.019807Z","shell.execute_reply":"2023-06-12T21:45:24.329389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:45:24.334688Z","iopub.execute_input":"2023-06-12T21:45:24.337005Z","iopub.status.idle":"2023-06-12T21:45:24.357527Z","shell.execute_reply.started":"2023-06-12T21:45:24.336968Z","shell.execute_reply":"2023-06-12T21:45:24.356227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(X, y, epochs=10, batch_size=32)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:45:24.359008Z","iopub.execute_input":"2023-06-12T21:45:24.359414Z","iopub.status.idle":"2023-06-12T21:45:59.552217Z","shell.execute_reply.started":"2023-06-12T21:45:24.359376Z","shell.execute_reply":"2023-06-12T21:45:59.551338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from matplotlib import pyplot as plt\nplt.figure(figsize=(8,5))\npd.DataFrame(history.history).plot()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:45:59.553936Z","iopub.execute_input":"2023-06-12T21:45:59.554584Z","iopub.status.idle":"2023-06-12T21:45:59.891251Z","shell.execute_reply.started":"2023-06-12T21:45:59.554543Z","shell.execute_reply":"2023-06-12T21:45:59.890114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_preds= model.predict(x_test)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:45:59.892771Z","iopub.execute_input":"2023-06-12T21:45:59.893784Z","iopub.status.idle":"2023-06-12T21:46:00.458667Z","shell.execute_reply.started":"2023-06-12T21:45:59.893723Z","shell.execute_reply":"2023-06-12T21:46:00.457358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_preds","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:46:00.461772Z","iopub.execute_input":"2023-06-12T21:46:00.462128Z","iopub.status.idle":"2023-06-12T21:46:00.469960Z","shell.execute_reply.started":"2023-06-12T21:46:00.462097Z","shell.execute_reply":"2023-06-12T21:46:00.468808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:46:00.471579Z","iopub.execute_input":"2023-06-12T21:46:00.472032Z","iopub.status.idle":"2023-06-12T21:46:00.484472Z","shell.execute_reply.started":"2023-06-12T21:46:00.471993Z","shell.execute_reply":"2023-06-12T21:46:00.483315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Vocabulary**","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.layers import TextVectorization\nMAX_FEATURES = 100000\n\nvectorizer = TextVectorization(max_tokens=MAX_FEATURES,\n                               output_sequence_length=1800,\n                               output_mode='int')","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:46:00.486173Z","iopub.execute_input":"2023-06-12T21:46:00.486549Z","iopub.status.idle":"2023-06-12T21:46:00.517975Z","shell.execute_reply.started":"2023-06-12T21:46:00.486516Z","shell.execute_reply":"2023-06-12T21:46:00.516990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df['question_text'][:10000]\ny = df['target'][:10000].values\n\nvectorizer.adapt(X.values)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:46:00.519881Z","iopub.execute_input":"2023-06-12T21:46:00.520214Z","iopub.status.idle":"2023-06-12T21:46:01.060953Z","shell.execute_reply.started":"2023-06-12T21:46:00.520186Z","shell.execute_reply":"2023-06-12T21:46:01.059862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vectorizer.get_vocabulary()","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:46:01.062222Z","iopub.execute_input":"2023-06-12T21:46:01.062995Z","iopub.status.idle":"2023-06-12T21:46:01.142535Z","shell.execute_reply.started":"2023-06-12T21:46:01.062964Z","shell.execute_reply":"2023-06-12T21:46:01.141456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vectorized_text = vectorizer(X.values)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:46:01.144240Z","iopub.execute_input":"2023-06-12T21:46:01.145098Z","iopub.status.idle":"2023-06-12T21:46:01.361035Z","shell.execute_reply.started":"2023-06-12T21:46:01.145059Z","shell.execute_reply":"2023-06-12T21:46:01.359677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(vectorized_text)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:46:01.362613Z","iopub.execute_input":"2023-06-12T21:46:01.362998Z","iopub.status.idle":"2023-06-12T21:46:01.369883Z","shell.execute_reply.started":"2023-06-12T21:46:01.362967Z","shell.execute_reply":"2023-06-12T21:46:01.368701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Logistic regression**","metadata":{}},{"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer\n\ndef clean_text(text):\n    text = text.lower()\n    text = re.sub(r\"[^a-zA-Z]\", \" \", text)\n    text = word_tokenize(text)\n    text = [word for word in text if word not in stop_words]\n    text = ' '.join(text)\n    return text\n\ntrain['cleaned_text'] = train['question_text'].apply(clean_text)\n\n\n# Text vectorization using TF-IDF\ntfidf = TfidfVectorizer()\nX = tfidf.fit_transform(train['cleaned_text'])\ny = train['target']\n\n# Splitting the data into train and validation sets\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Train the model\nlr = LogisticRegression()\nlr.fit(X_train, y_train)\n\n# Make predictions\ntrain_pred = lr.predict(X_train)\nval_pred = lr.predict(X_val)\n\n# Evaluation metrics\ntrain_acc = accuracy_score(y_train, train_pred)\nval_acc = accuracy_score(y_val, val_pred)\n\nprint(\"Training Accuracy:\", train_acc)\nprint(\"Validation Accuracy:\", val_acc)\n\n# Show some correct and incorrect predictions\ndef show_examples(predictions, actual_labels, text_data, num_examples=5):\n    correct_indices = np.where(predictions == actual_labels)[0]\n    incorrect_indices = np.where(predictions != actual_labels)[0]\n\n    print(\"Correct Predictions:\")\n    for idx in np.random.choice(correct_indices, size=num_examples, replace=False):\n        print(\"Question:\", text_data.iloc[idx])\n        print(\"Prediction:\", predictions[idx])\n        print(\"Actual Label:\", actual_labels.iloc[idx])\n        print()\n\n    print(\"Incorrect Predictions:\")\n    for idx in np.random.choice(incorrect_indices, size=num_examples, replace=False):\n        print(\"Question:\", text_data.iloc[idx])\n        print(\"Prediction:\", predictions[idx])\n        print(\"Actual Label:\", actual_labels.iloc[idx])\n        print()\n\nshow_examples(val_pred, y_val, train['question_text'])","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:46:01.375620Z","iopub.execute_input":"2023-06-12T21:46:01.376125Z","iopub.status.idle":"2023-06-12T21:51:23.621602Z","shell.execute_reply.started":"2023-06-12T21:46:01.376084Z","shell.execute_reply":"2023-06-12T21:51:23.620415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"disp = metrics.ConfusionMatrixDisplay.from_predictions(y_val, val_pred)\ndisp.figure_.suptitle(\"Confusion Matrix\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:51:23.622891Z","iopub.execute_input":"2023-06-12T21:51:23.623258Z","iopub.status.idle":"2023-06-12T21:51:23.947798Z","shell.execute_reply.started":"2023-06-12T21:51:23.623230Z","shell.execute_reply":"2023-06-12T21:51:23.946732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\nfrom nltk.stem.snowball import SnowballStemmer\nstemmer = SnowballStemmer(language='english')\nfrom nltk.corpus import stopwords\nnltk.download('stopwords')\nenglish_stopwords = stopwords.words('english')\n\ntest = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/test.csv')\n\n\ndef tokenize(text):\n  return [stemmer.stem(word) for word in word_tokenize(text) if word.lower() not in english_stopwords]\n\nvectorizer = CountVectorizer(lowercase=True, tokenizer=tokenize, stop_words=english_stopwords, max_features=1000)\n\nvectorizer.fit(df['question_text'])\ninputs = vectorizer.transform(df['question_text'])\ntest_inputs = vectorizer.transform(test['question_text'])\n","metadata":{"execution":{"iopub.status.busy":"2023-06-12T21:57:42.320505Z","iopub.execute_input":"2023-06-12T21:57:42.320994Z","iopub.status.idle":"2023-06-12T22:00:30.963908Z","shell.execute_reply.started":"2023-06-12T21:57:42.320962Z","shell.execute_reply":"2023-06-12T22:00:30.962690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_inputs, val_inputs, train_targets, val_targets = train_test_split(inputs, df['target'], test_size=0.3, random_state=42)\n\n\nmodel = LogisticRegression(max_iter=1000, solver='sag')\nmodel.fit(train_inputs, train_targets)\n\n\ntrain_preds = model.predict(train_inputs)\nval_preds= model.predict(val_inputs)\n\n\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import f1_score\n\nprint(f\"Training Accuracy {accuracy_score(train_targets, train_preds)}\")\nprint(f\"Training F1 Score {f1_score(train_targets, train_preds)}\")\n\nprint(f\"Validation Accuracy {accuracy_score(val_targets, val_preds)}\")\nprint(f\"Validation F1 Score {f1_score(val_targets, val_preds)}\")\n\n\ntest_preds = model.predict(test_inputs)\n\nsubmission_df = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/sample_submission.csv\")\n\nsubmission_df['prediction'] = test_preds\nsubmission_df['prediction'].value_counts()\n\nsubmission_df.to_csv('submission.csv', index=None)\ntest_pred_csv = pd.read_csv(\"/kaggle/working/submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-06-12T22:00:30.966206Z","iopub.execute_input":"2023-06-12T22:00:30.966579Z","iopub.status.idle":"2023-06-12T22:00:32.802459Z","shell.execute_reply.started":"2023-06-12T22:00:30.966549Z","shell.execute_reply":"2023-06-12T22:00:32.801217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"disp = metrics.ConfusionMatrixDisplay.from_predictions(val_targets, val_preds)\ndisp.figure_.suptitle(\"Confusion Matrix\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-12T22:00:32.803999Z","iopub.execute_input":"2023-06-12T22:00:32.804911Z","iopub.status.idle":"2023-06-12T22:00:33.081734Z","shell.execute_reply.started":"2023-06-12T22:00:32.804878Z","shell.execute_reply":"2023-06-12T22:00:33.080600Z"},"trusted":true},"execution_count":null,"outputs":[]}]}