{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":10737,"databundleVersionId":290346,"sourceType":"competition"}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-18T05:22:32.787903Z","iopub.execute_input":"2024-04-18T05:22:32.788329Z","iopub.status.idle":"2024-04-18T05:22:34.181356Z","shell.execute_reply.started":"2024-04-18T05:22:32.788294Z","shell.execute_reply":"2024-04-18T05:22:34.180475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")\nimport numpy as np \nimport pandas as pd \nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport nltk\nfrom sklearn.pipeline import Pipeline\nfrom nltk.corpus import stopwords\nfrom string import punctuation \nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.feature_extraction.text import CountVectorizer,TfidfVectorizer\nfrom sklearn import model_selection\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import f1_score\nfrom wordcloud import WordCloud, STOPWORDS\nfrom nltk.corpus import stopwords \nimport nltk\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize\nimport string\nfrom sklearn.model_selection import train_test_split\nimport nltk\nfrom nltk.corpus import stopwords\nimport pandas as pd\nfrom imblearn.under_sampling import RandomUnderSampler\nimport pandas as pd\nfrom sklearn.metrics import confusion_matrix, classification_report\nnltk.download('stopwords')\nnltk.download('punkt')","metadata":{"execution":{"iopub.status.busy":"2024-04-18T05:25:59.374226Z","iopub.execute_input":"2024-04-18T05:25:59.374807Z","iopub.status.idle":"2024-04-18T05:26:02.936992Z","shell.execute_reply.started":"2024-04-18T05:25:59.374775Z","shell.execute_reply":"2024-04-18T05:26:02.93531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/train.csv\")\ndf_test  = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-04-18T05:26:23.821928Z","iopub.execute_input":"2024-04-18T05:26:23.823457Z","iopub.status.idle":"2024-04-18T05:26:30.836674Z","shell.execute_reply.started":"2024-04-18T05:26:23.823403Z","shell.execute_reply":"2024-04-18T05:26:30.835282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-18T05:26:48.495624Z","iopub.execute_input":"2024-04-18T05:26:48.496031Z","iopub.status.idle":"2024-04-18T05:26:48.51789Z","shell.execute_reply.started":"2024-04-18T05:26:48.496002Z","shell.execute_reply":"2024-04-18T05:26:48.516608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-18T05:27:04.239783Z","iopub.execute_input":"2024-04-18T05:27:04.24021Z","iopub.status.idle":"2024-04-18T05:27:04.255278Z","shell.execute_reply.started":"2024-04-18T05:27:04.240179Z","shell.execute_reply":"2024-04-18T05:27:04.253528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[\"target\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-04-18T05:27:21.68593Z","iopub.execute_input":"2024-04-18T05:27:21.686326Z","iopub.status.idle":"2024-04-18T05:27:21.715793Z","shell.execute_reply.started":"2024-04-18T05:27:21.686296Z","shell.execute_reply":"2024-04-18T05:27:21.714624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"question_class = df_train[\"target\"].value_counts()\ncolors = [\"#B92B27\", \"dodgerblue\"]\nquestion_class.plot(kind=\"bar\", color=colors, edgecolor=\"black\")\nplt.xlabel(\"Question Class\")\nplt.ylabel(\"Number of Questions\")\nplt.title(\"Distribution of Question Classes in Training Data\")\nfor i, count in enumerate(question_class):\n    plt.text(i, count + 0.1, str(count), ha=\"center\", va=\"bottom\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-18T05:27:42.106051Z","iopub.execute_input":"2024-04-18T05:27:42.106519Z","iopub.status.idle":"2024-04-18T05:27:42.466904Z","shell.execute_reply.started":"2024-04-18T05:27:42.106485Z","shell.execute_reply":"2024-04-18T05:27:42.465734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_train[\"target\"].value_counts())\nprint(sum(df_train[\"target\"] == 1) / sum(df_train[\"target\"] == 0) * 100, \"percent of questions are insincere.\")\nprint(100 - sum(df_train[\"target\"] == 1) / sum(df_train[\"target\"] == 0) * 100, \"percent of questions are sincere\")","metadata":{"execution":{"iopub.status.busy":"2024-04-18T05:28:21.121918Z","iopub.execute_input":"2024-04-18T05:28:21.122673Z","iopub.status.idle":"2024-04-18T05:28:21.865184Z","shell.execute_reply.started":"2024-04-18T05:28:21.122636Z","shell.execute_reply":"2024-04-18T05:28:21.863976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Original Class Distribution:\")\nprint(df_train[\"target\"].value_counts())\n\nrus = RandomUnderSampler(random_state=42)\nX_resampled, y_resampled = rus.fit_resample(df_train.drop(\"target\", axis=1), df_train[\"target\"])\n\ndf_balanced = pd.concat([X_resampled, y_resampled], axis=1)\n\nprint(\"\\nBalanced Class Distribution:\")\nprint(df_balanced[\"target\"].value_counts())\n\ninsincere_percent = sum(df_balanced[\"target\"] == 1) / len(df_balanced) * 100\nsincere_percent = 100 - insincere_percent\n\nprint(f\"\\n{insincere_percent:.2f}% of questions are insincere.\")\nprint(f\"{sincere_percent:.2f}% of questions are sincere.\")","metadata":{"execution":{"iopub.status.busy":"2024-04-18T05:28:36.449596Z","iopub.execute_input":"2024-04-18T05:28:36.450853Z","iopub.status.idle":"2024-04-18T05:28:37.063191Z","shell.execute_reply.started":"2024-04-18T05:28:36.450806Z","shell.execute_reply":"2024-04-18T05:28:37.061952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_balanced.copy()\ndf_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-18T05:29:34.729259Z","iopub.execute_input":"2024-04-18T05:29:34.730165Z","iopub.status.idle":"2024-04-18T05:29:34.747405Z","shell.execute_reply.started":"2024-04-18T05:29:34.730124Z","shell.execute_reply":"2024-04-18T05:29:34.746055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[\"number_words\"] = df_train[\"question_text\"].apply(lambda x: len(x.split()))\ndf_test[\"number_words\"]  = df_test[\"question_text\"].apply(lambda x: len(x.split()))","metadata":{"execution":{"iopub.status.busy":"2024-04-18T05:34:41.016719Z","iopub.execute_input":"2024-04-18T05:34:41.017095Z","iopub.status.idle":"2024-04-18T05:34:41.993927Z","shell.execute_reply.started":"2024-04-18T05:34:41.017066Z","shell.execute_reply":"2024-04-18T05:34:41.992972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[\"num_chars\"] = df_train[\"question_text\"].apply(lambda x: len(str(x)))\ndf_test[\"num_chars\"]  = df_test[\"question_text\"].apply(lambda x: len(str(x)))","metadata":{"execution":{"iopub.status.busy":"2024-04-18T05:37:35.594869Z","iopub.execute_input":"2024-04-18T05:37:35.595613Z","iopub.status.idle":"2024-04-18T05:37:36.10316Z","shell.execute_reply.started":"2024-04-18T05:37:35.595577Z","shell.execute_reply":"2024-04-18T05:37:36.101812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import nltk\nfrom nltk.corpus import stopwords\nimport pandas as pd\nnltk.download('stopwords')\nstop_words = set(stopwords.words(\"english\"))\ndf_train[\"num_stopwords\"] = df_train[\"question_text\"].apply(lambda x : len([nw for nw in str(x).split() if nw.lower() in stop_words]))\ndf_test[\"num_stopwords\"]  = df_test[\"question_text\"].apply(lambda x : len([nw for nw in str(x).split() if nw.lower() in stop_words]))","metadata":{"execution":{"iopub.status.busy":"2024-04-18T05:49:59.425075Z","iopub.execute_input":"2024-04-18T05:49:59.425974Z","iopub.status.idle":"2024-04-18T05:50:02.500868Z","shell.execute_reply.started":"2024-04-18T05:49:59.425931Z","shell.execute_reply":"2024-04-18T05:50:02.499733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[\"num_punctuation\"] = df_train[\"question_text\"].apply(lambda x : len([np for np in str(x) if np in punctuation]))\ndf_test[\"num_punctuation\"]  = df_test[\"question_text\"].apply(lambda x : len([np for np in str(x) if np in punctuation]))","metadata":{"execution":{"iopub.status.busy":"2024-04-18T05:53:43.109491Z","iopub.execute_input":"2024-04-18T05:53:43.109905Z","iopub.status.idle":"2024-04-18T05:53:46.614787Z","shell.execute_reply.started":"2024-04-18T05:53:43.109876Z","shell.execute_reply":"2024-04-18T05:53:46.613642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[\"num_uppercase\"] = df_train[\"question_text\"].apply(lambda x : len([nu for nu in str(x).split() if nu.isupper()]))\ndf_test[\"num_uppercase\"]  = df_test[\"question_text\"].apply(lambda x : len([nu for nu in str(x).split() if nu.isupper()]))","metadata":{"execution":{"iopub.status.busy":"2024-04-18T05:55:34.794927Z","iopub.execute_input":"2024-04-18T05:55:34.795348Z","iopub.status.idle":"2024-04-18T05:55:36.747768Z","shell.execute_reply.started":"2024-04-18T05:55:34.795318Z","shell.execute_reply":"2024-04-18T05:55:36.745946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[\"num_lowercase\"] = df_train[\"question_text\"].apply(lambda x : len([nl for nl in str(x).split() if nl.islower()]))\ndf_test[\"num_lowercase\"]  = df_test[\"question_text\"].apply(lambda x : len([nl for nl in str(x).split() if nl.islower()]))","metadata":{"execution":{"iopub.status.busy":"2024-04-18T05:56:46.520815Z","iopub.execute_input":"2024-04-18T05:56:46.521256Z","iopub.status.idle":"2024-04-18T05:56:49.14944Z","shell.execute_reply.started":"2024-04-18T05:56:46.521225Z","shell.execute_reply":"2024-04-18T05:56:49.148431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[\"num_unique_words\"] = df_train[\"question_text\"].apply(lambda x: len(set(str(x).split())))\ndf_test[\"num_unique_words\"]  = df_test[\"question_text\"].apply(lambda x: len(set(str(x).split())))","metadata":{"execution":{"iopub.status.busy":"2024-04-18T06:18:58.798197Z","iopub.execute_input":"2024-04-18T06:18:58.798657Z","iopub.status.idle":"2024-04-18T06:19:00.79806Z","shell.execute_reply.started":"2024-04-18T06:18:58.798624Z","shell.execute_reply":"2024-04-18T06:19:00.796711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\ndf_train[\"num_title\"] = df_train[\"question_text\"].apply(lambda x : len([nl for nl in str(x).split() if nl.istitle()]))\ndf_test[\"num_title\"]  = df_test[\"question_text\"].apply(lambda x : len([nl for nl in str(x).split() if nl.istitle()]))\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-18T06:48:47.574534Z","iopub.execute_input":"2024-04-18T06:48:47.574995Z","iopub.status.idle":"2024-04-18T06:48:49.830202Z","shell.execute_reply.started":"2024-04-18T06:48:47.574959Z","shell.execute_reply":"2024-04-18T06:48:49.829228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[df_train[\"target\"] == 1].describe()","metadata":{"execution":{"iopub.status.busy":"2024-04-18T06:19:08.385296Z","iopub.execute_input":"2024-04-18T06:19:08.386216Z","iopub.status.idle":"2024-04-18T06:19:08.485831Z","shell.execute_reply.started":"2024-04-18T06:19:08.38616Z","shell.execute_reply":"2024-04-18T06:19:08.484397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[df_train[\"target\"] == 0].describe()","metadata":{"execution":{"iopub.status.busy":"2024-04-18T06:19:09.225326Z","iopub.execute_input":"2024-04-18T06:19:09.225768Z","iopub.status.idle":"2024-04-18T06:19:09.303657Z","shell.execute_reply.started":"2024-04-18T06:19:09.225734Z","shell.execute_reply":"2024-04-18T06:19:09.302455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(12, 10))\ncolors = [\"#B92B27\", \"dodgerblue\"]\nsns.set_palette(sns.color_palette(colors))\nsns.boxplot(data=df_train, y=\"number_words\", x=\"target\", orient=\"v\", ax=ax)\nax.set(xlabel=\"Target\", ylabel=\"Number of Words\", title=\"Box Plot of Number of Words According to the Target\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-18T06:19:09.672464Z","iopub.execute_input":"2024-04-18T06:19:09.672883Z","iopub.status.idle":"2024-04-18T06:19:10.039131Z","shell.execute_reply.started":"2024-04-18T06:19:09.67285Z","shell.execute_reply":"2024-04-18T06:19:10.037279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(12, 10))\ncolors = [\"#B92B27\", \"dodgerblue\"]\nsns.set_palette(sns.color_palette(colors))\nsns.boxplot(data=df_train, y=\"num_unique_words\", x=\"target\", orient=\"v\", ax=ax)\nax.set(xlabel=\"Target\", ylabel=\"Number of Unique Words\", title=\"Box Plot of Number of Unique Words According to the Target\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-18T06:19:10.597707Z","iopub.execute_input":"2024-04-18T06:19:10.598117Z","iopub.status.idle":"2024-04-18T06:19:10.955743Z","shell.execute_reply.started":"2024-04-18T06:19:10.598086Z","shell.execute_reply":"2024-04-18T06:19:10.95445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(12, 10))\ncolors = [\"#B92B27\", \"dodgerblue\"]\nsns.set_palette(sns.color_palette(colors))\nsns.boxplot(data=df_train, y=\"num_chars\", x=\"target\", orient=\"v\", ax=ax)\nax.set(xlabel=\"Target\", ylabel=\"Number of Chars\", title=\"Box Plot of Number of Chars According to the Target\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-18T06:24:12.170471Z","iopub.execute_input":"2024-04-18T06:24:12.171531Z","iopub.status.idle":"2024-04-18T06:24:12.523781Z","shell.execute_reply.started":"2024-04-18T06:24:12.171492Z","shell.execute_reply":"2024-04-18T06:24:12.522451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(12, 10))\ncolors = [\"#B92B27\", \"dodgerblue\"]\nsns.set_palette(sns.color_palette(colors))\nsns.boxplot(data=df_train, y=\"num_stopwords\", x=\"target\", orient=\"v\", ax=ax)\nax.set(xlabel=\"Target\", ylabel=\"Number of Stop Words\", title=\"Box Plot of Number of Stop Words According to the Target\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-18T06:24:38.362058Z","iopub.execute_input":"2024-04-18T06:24:38.362529Z","iopub.status.idle":"2024-04-18T06:24:38.763431Z","shell.execute_reply.started":"2024-04-18T06:24:38.362493Z","shell.execute_reply":"2024-04-18T06:24:38.76223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(12, 10))\ncolors = [\"#B92B27\", \"dodgerblue\"]\nsns.set_palette(sns.color_palette(colors))\nsns.boxplot(data=df_train, y=\"num_lowercase\", x=\"target\", orient=\"v\", ax=ax)\nax.set(xlabel=\"Target\", ylabel=\"Number of LowerCase Words\", title=\"Box Plot of Number of LowerCase Words According to the Target\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-18T06:47:58.227997Z","iopub.execute_input":"2024-04-18T06:47:58.228537Z","iopub.status.idle":"2024-04-18T06:47:58.624694Z","shell.execute_reply.started":"2024-04-18T06:47:58.228499Z","shell.execute_reply":"2024-04-18T06:47:58.623497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nfig, ax = plt.subplots(figsize=(12, 10))\ncolors = [\"#B92B27\", \"dodgerblue\"]\nsns.set_palette(sns.color_palette(colors))\nsns.boxplot(data=df_train, y=\"num_title\", x=\"target\", orient=\"v\", ax=ax)\nax.set(xlabel=\"Target\", ylabel=\"Number of Titles\", title=\"Box Plot of Number of Titles According to the Target\")\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-18T06:48:55.904303Z","iopub.execute_input":"2024-04-18T06:48:55.904975Z","iopub.status.idle":"2024-04-18T06:48:56.288025Z","shell.execute_reply.started":"2024-04-18T06:48:55.90494Z","shell.execute_reply":"2024-04-18T06:48:56.286858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\ndf_train[df_train[\"target\"] == 1].describe()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-18T06:49:13.858074Z","iopub.execute_input":"2024-04-18T06:49:13.858513Z","iopub.status.idle":"2024-04-18T06:49:13.950171Z","shell.execute_reply.started":"2024-04-18T06:49:13.858478Z","shell.execute_reply":"2024-04-18T06:49:13.948984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\ndf_train[df_train[\"target\"] == 0].describe()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-18T06:49:39.759331Z","iopub.execute_input":"2024-04-18T06:49:39.760363Z","iopub.status.idle":"2024-04-18T06:49:39.85216Z","shell.execute_reply.started":"2024-04-18T06:49:39.760306Z","shell.execute_reply":"2024-04-18T06:49:39.850823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(12, 10))\ncolors = [\"#B92B27\", \"dodgerblue\"]\nsns.set_palette(sns.color_palette(colors))\nsns.boxplot(data=df_train, y=\"number_words\", x=\"target\", orient=\"v\", ax=ax)\nax.set(xlabel=\"Target\", ylabel=\"Number of Words\", title=\"Box Plot of Number of Words According to the Target\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-18T06:50:20.020765Z","iopub.execute_input":"2024-04-18T06:50:20.021209Z","iopub.status.idle":"2024-04-18T06:50:20.410143Z","shell.execute_reply.started":"2024-04-18T06:50:20.021179Z","shell.execute_reply":"2024-04-18T06:50:20.408737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nfig, ax = plt.subplots(figsize=(12, 10))\ncolors = [\"#B92B27\", \"dodgerblue\"]\nsns.set_palette(sns.color_palette(colors))\nsns.boxplot(data=df_train, y=\"num_unique_words\", x=\"target\", orient=\"v\", ax=ax)\nax.set(xlabel=\"Target\", ylabel=\"Number of Unique Words\", title=\"Box Plot of Number of Unique Words According to the Target\")\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-18T06:50:29.838748Z","iopub.execute_input":"2024-04-18T06:50:29.839194Z","iopub.status.idle":"2024-04-18T06:50:30.200085Z","shell.execute_reply.started":"2024-04-18T06:50:29.839159Z","shell.execute_reply":"2024-04-18T06:50:30.198652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(12, 10))\ncolors = [\"#B92B27\", \"dodgerblue\"]\nsns.set_palette(sns.color_palette(colors))\nsns.boxplot(data=df_train, y=\"num_chars\", x=\"target\", orient=\"v\", ax=ax)\nax.set(xlabel=\"Target\", ylabel=\"Number of Chars\", title=\"Box Plot of Number of Chars According to the Target\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-18T06:50:39.278484Z","iopub.execute_input":"2024-04-18T06:50:39.278952Z","iopub.status.idle":"2024-04-18T06:50:39.650209Z","shell.execute_reply.started":"2024-04-18T06:50:39.278919Z","shell.execute_reply":"2024-04-18T06:50:39.648807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(12, 10))\ncolors = [\"#B92B27\", \"dodgerblue\"]\nsns.set_palette(sns.color_palette(colors))\nsns.boxplot(data=df_train, y=\"num_stopwords\", x=\"target\", orient=\"v\", ax=ax)\nax.set(xlabel=\"Target\", ylabel=\"Number of Stop Words\", title=\"Box Plot of Number of Stop Words According to the Target\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-18T06:50:48.89159Z","iopub.execute_input":"2024-04-18T06:50:48.892005Z","iopub.status.idle":"2024-04-18T06:50:49.296005Z","shell.execute_reply.started":"2024-04-18T06:50:48.891974Z","shell.execute_reply":"2024-04-18T06:50:49.294682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def text_process(text):\n    text = text.lower()\n    text = text.translate(str.maketrans(\"\", \"\", string.punctuation))\n    tokens = word_tokenize(text)\n    stop_words = set(stopwords.words('english'))\n    tokens = [word for word in tokens if word not in stop_words]\n    processed_text = ' '.join(tokens)\n    return processed_text","metadata":{"execution":{"iopub.status.busy":"2024-04-18T06:51:08.398682Z","iopub.execute_input":"2024-04-18T06:51:08.399647Z","iopub.status.idle":"2024-04-18T06:51:08.411163Z","shell.execute_reply.started":"2024-04-18T06:51:08.399595Z","shell.execute_reply":"2024-04-18T06:51:08.409475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['clean_train'] = df_train[\"question_text\"].apply(text_process)\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-18T06:51:18.319743Z","iopub.execute_input":"2024-04-18T06:51:18.320154Z","iopub.status.idle":"2024-04-18T06:52:28.662694Z","shell.execute_reply.started":"2024-04-18T06:51:18.320123Z","shell.execute_reply":"2024-04-18T06:52:28.660761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\ndf_test['clean_test'] = df_test[\"question_text\"].apply(text_process)\ndf_test\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-18T06:52:57.549051Z","iopub.execute_input":"2024-04-18T06:52:57.549819Z","iopub.status.idle":"2024-04-18T06:55:32.524735Z","shell.execute_reply.started":"2024-04-18T06:52:57.549765Z","shell.execute_reply":"2024-04-18T06:55:32.523438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train,X_val,Y_train,Y_val = train_test_split(df_train['clean_train'],df_train['target'],test_size=0.2)\nX_train.shape,X_val.shape,Y_train.shape,Y_val.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-18T06:56:22.25247Z","iopub.execute_input":"2024-04-18T06:56:22.25331Z","iopub.status.idle":"2024-04-18T06:56:22.291539Z","shell.execute_reply.started":"2024-04-18T06:56:22.253269Z","shell.execute_reply":"2024-04-18T06:56:22.290192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\npipeline = Pipeline(\n    [\n        (\"cv\",CountVectorizer(analyzer=\"word\",ngram_range=(1,4),max_df=0.9)),\n        (\"clf\",LogisticRegression(solver=\"saga\", class_weight=\"balanced\", C=0.45, max_iter=250, verbose=1))\n    ]\n)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-18T06:56:33.244183Z","iopub.execute_input":"2024-04-18T06:56:33.244895Z","iopub.status.idle":"2024-04-18T06:56:33.252215Z","shell.execute_reply.started":"2024-04-18T06:56:33.244859Z","shell.execute_reply":"2024-04-18T06:56:33.250266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nlr_model = pipeline.fit(X_train,Y_train)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-18T06:56:43.032237Z","iopub.execute_input":"2024-04-18T06:56:43.032686Z","iopub.status.idle":"2024-04-18T06:57:42.413751Z","shell.execute_reply.started":"2024-04-18T06:56:43.032654Z","shell.execute_reply":"2024-04-18T06:57:42.412379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_pred = lr_model.predict(X_val)\nprint(classification_report(Y_val,Y_pred))\ncm     = confusion_matrix(Y_val,Y_pred)\nsns.heatmap(cm, cmap=\"Blues\", annot=True, square=True, fmt=\".0f\");","metadata":{"execution":{"iopub.status.busy":"2024-04-18T07:05:42.282833Z","iopub.execute_input":"2024-04-18T07:05:42.283428Z","iopub.status.idle":"2024-04-18T07:05:44.168827Z","shell.execute_reply.started":"2024-04-18T07:05:42.283387Z","shell.execute_reply":"2024-04-18T07:05:44.167062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_final = pipeline.predict(df_test['clean_test'])\ny_pred_final","metadata":{"execution":{"iopub.status.busy":"2024-04-18T07:06:35.926953Z","iopub.execute_input":"2024-04-18T07:06:35.927996Z","iopub.status.idle":"2024-04-18T07:06:49.605344Z","shell.execute_reply.started":"2024-04-18T07:06:35.927957Z","shell.execute_reply":"2024-04-18T07:06:49.604047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub = pd.DataFrame({\"qid\":df_test[\"qid\"], \"prediction\":y_pred_final})\ndf_sub.to_csv('submission.csv', index=False)\ndf_sub.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-18T07:06:49.607868Z","iopub.execute_input":"2024-04-18T07:06:49.608449Z","iopub.status.idle":"2024-04-18T07:06:50.487605Z","shell.execute_reply.started":"2024-04-18T07:06:49.608409Z","shell.execute_reply":"2024-04-18T07:06:50.486266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}