{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":10737,"databundleVersionId":290346,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-10-25T03:08:00.642503Z","iopub.execute_input":"2024-10-25T03:08:00.642981Z","iopub.status.idle":"2024-10-25T03:08:02.490349Z","shell.execute_reply.started":"2024-10-25T03:08:00.642934Z","shell.execute_reply":"2024-10-25T03:08:02.488827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/train.csv')\ntrain","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-25T03:08:09.100224Z","iopub.execute_input":"2024-10-25T03:08:09.100861Z","iopub.status.idle":"2024-10-25T03:08:14.252475Z","shell.execute_reply.started":"2024-10-25T03:08:09.100797Z","shell.execute_reply":"2024-10-25T03:08:14.251139Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/test.csv')\ntest","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-25T03:08:15.719049Z","iopub.execute_input":"2024-10-25T03:08:15.719534Z","iopub.status.idle":"2024-10-25T03:08:16.746928Z","shell.execute_reply.started":"2024-10-25T03:08:15.719487Z","shell.execute_reply":"2024-10-25T03:08:16.745570Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-25T03:08:16.749101Z","iopub.execute_input":"2024-10-25T03:08:16.749510Z","iopub.status.idle":"2024-10-25T03:08:17.086991Z","shell.execute_reply.started":"2024-10-25T03:08:16.749465Z","shell.execute_reply":"2024-10-25T03:08:17.085432Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-25T03:08:17.311674Z","iopub.execute_input":"2024-10-25T03:08:17.312124Z","iopub.status.idle":"2024-10-25T03:08:17.408004Z","shell.execute_reply.started":"2024-10-25T03:08:17.312079Z","shell.execute_reply":"2024-10-25T03:08:17.406650Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.duplicated().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-25T03:08:19.521355Z","iopub.execute_input":"2024-10-25T03:08:19.521860Z","iopub.status.idle":"2024-10-25T03:08:20.717651Z","shell.execute_reply.started":"2024-10-25T03:08:19.521795Z","shell.execute_reply":"2024-10-25T03:08:20.716590Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.duplicated().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-25T03:08:21.663092Z","iopub.execute_input":"2024-10-25T03:08:21.663541Z","iopub.status.idle":"2024-10-25T03:08:21.989957Z","shell.execute_reply.started":"2024-10-25T03:08:21.663496Z","shell.execute_reply":"2024-10-25T03:08:21.988744Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"There are no null and duplicated values!","metadata":{}},{"cell_type":"code","source":"train['target'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-25T03:08:23.813359Z","iopub.execute_input":"2024-10-25T03:08:23.813799Z","iopub.status.idle":"2024-10-25T03:08:23.839485Z","shell.execute_reply.started":"2024-10-25T03:08:23.813757Z","shell.execute_reply":"2024-10-25T03:08:23.838122Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The data is highly imbalanced","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-25T03:08:25.819078Z","iopub.execute_input":"2024-10-25T03:08:25.819505Z","iopub.status.idle":"2024-10-25T03:08:26.818154Z","shell.execute_reply.started":"2024-10-25T03:08:25.819464Z","shell.execute_reply":"2024-10-25T03:08:26.816601Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ax = sns.barplot(x = train['target'].value_counts().index, y = train['target'].value_counts().values)\nplt.title('Distribution of target variable')\nax.set_xticklabels(['Insincere', 'Sincere'])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-25T03:08:27.657344Z","iopub.execute_input":"2024-10-25T03:08:27.658002Z","iopub.status.idle":"2024-10-25T03:08:27.997703Z","shell.execute_reply.started":"2024-10-25T03:08:27.657950Z","shell.execute_reply":"2024-10-25T03:08:27.995955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Counting the number of words in the text column\ntrain['text_len'] = train['question_text'].apply(lambda x:len(x.split()))\ntrain","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-25T03:08:31.441532Z","iopub.execute_input":"2024-10-25T03:08:31.442599Z","iopub.status.idle":"2024-10-25T03:08:33.736110Z","shell.execute_reply.started":"2024-10-25T03:08:31.442527Z","shell.execute_reply":"2024-10-25T03:08:33.734716Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.histplot(data = train, x = 'text_len', bins = 20, kde = True)\nplt.title('Histogram of number of words in a text')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-25T03:08:34.079524Z","iopub.execute_input":"2024-10-25T03:08:34.080025Z","iopub.status.idle":"2024-10-25T03:08:41.045894Z","shell.execute_reply.started":"2024-10-25T03:08:34.079975Z","shell.execute_reply":"2024-10-25T03:08:41.044539Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"As, we can see most of the sentences here have around 10 words","metadata":{}},{"cell_type":"code","source":"v = train['question_text'][1]\nv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-25T03:08:41.048297Z","iopub.execute_input":"2024-10-25T03:08:41.048759Z","iopub.status.idle":"2024-10-25T03:08:41.059344Z","shell.execute_reply.started":"2024-10-25T03:08:41.048706Z","shell.execute_reply":"2024-10-25T03:08:41.058045Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def most_frequent_words(df, col):\n    d = {}\n    for row in df[col]:\n        for word in row.split():\n            if word in d:\n                d[word] = d[word] + 1\n            else:\n                d[word] = 1\n    t = sorted(d.items(), key=lambda x:-x[1])[:20]\n    print(t)\n    most_freq_words = pd.DataFrame(t, columns = ['words','count'])\n    sns.barplot(data = most_freq_words, x = 'count', y = 'words', color= 'skyblue')\n    plt.title(\"Top 20 most frequent words\")\n    plt.show()\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-25T03:27:23.140377Z","iopub.execute_input":"2024-10-25T03:27:23.140891Z","iopub.status.idle":"2024-10-25T03:27:23.150692Z","shell.execute_reply.started":"2024-10-25T03:27:23.140817Z","shell.execute_reply":"2024-10-25T03:27:23.149093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"most_frequent_words(train, 'question_text')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-25T03:27:23.959624Z","iopub.execute_input":"2024-10-25T03:27:23.960758Z","iopub.status.idle":"2024-10-25T03:27:33.473682Z","shell.execute_reply.started":"2024-10-25T03:27:23.960692Z","shell.execute_reply":"2024-10-25T03:27:33.472450Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"most_frequent_words(test, 'question_text')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-25T03:28:16.600095Z","iopub.execute_input":"2024-10-25T03:28:16.600551Z","iopub.status.idle":"2024-10-25T03:28:19.631885Z","shell.execute_reply.started":"2024-10-25T03:28:16.600505Z","shell.execute_reply":"2024-10-25T03:28:19.630557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}