{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":10737,"databundleVersionId":290346,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-11T04:16:42.99995Z","iopub.execute_input":"2024-03-11T04:16:43.000273Z","iopub.status.idle":"2024-03-11T04:16:43.375705Z","shell.execute_reply.started":"2024-03-11T04:16:43.000246Z","shell.execute_reply":"2024-03-11T04:16:43.374507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Importing Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport string\nimport re\nimport nltk\n\nfrom tqdm import trange\nfrom nltk import tokenize\nfrom nltk.corpus import stopwords\nfrom nltk.stem import WordNetLemmatizer\nfrom nltk.probability import FreqDist\nfrom collections import Counter\nfrom sklearn.feature_extraction.text import CountVectorizer","metadata":{"execution":{"iopub.status.busy":"2024-03-11T04:16:46.42312Z","iopub.execute_input":"2024-03-11T04:16:46.423621Z","iopub.status.idle":"2024-03-11T04:16:47.891643Z","shell.execute_reply.started":"2024-03-11T04:16:46.423593Z","shell.execute_reply":"2024-03-11T04:16:47.890345Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp_df=pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-03-11T04:16:50.300353Z","iopub.execute_input":"2024-03-11T04:16:50.300715Z","iopub.status.idle":"2024-03-11T04:16:54.048466Z","shell.execute_reply.started":"2024-03-11T04:16:50.300687Z","shell.execute_reply":"2024-03-11T04:16:54.047198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=temp_df.iloc[:20000].copy()\ndf.drop(columns=['qid'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T04:16:54.050364Z","iopub.execute_input":"2024-03-11T04:16:54.050706Z","iopub.status.idle":"2024-03-11T04:16:54.067201Z","shell.execute_reply.started":"2024-03-11T04:16:54.050682Z","shell.execute_reply":"2024-03-11T04:16:54.066307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2024-03-11T04:16:54.068348Z","iopub.execute_input":"2024-03-11T04:16:54.068624Z","iopub.status.idle":"2024-03-11T04:16:54.100463Z","shell.execute_reply.started":"2024-03-11T04:16:54.068601Z","shell.execute_reply":"2024-03-11T04:16:54.099317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-11T04:16:54.102174Z","iopub.execute_input":"2024-03-11T04:16:54.102864Z","iopub.status.idle":"2024-03-11T04:16:54.114139Z","shell.execute_reply.started":"2024-03-11T04:16:54.102839Z","shell.execute_reply":"2024-03-11T04:16:54.113507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:01:13.615Z","iopub.execute_input":"2024-03-11T05:01:13.615346Z","iopub.status.idle":"2024-03-11T05:01:13.62229Z","shell.execute_reply.started":"2024-03-11T05:01:13.61532Z","shell.execute_reply":"2024-03-11T05:01:13.621133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['target'].value_counts()\n","metadata":{"execution":{"iopub.status.busy":"2024-03-11T04:16:55.029159Z","iopub.execute_input":"2024-03-11T04:16:55.029696Z","iopub.status.idle":"2024-03-11T04:16:55.037913Z","shell.execute_reply.started":"2024-03-11T04:16:55.02967Z","shell.execute_reply":"2024-03-11T04:16:55.037268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ratings(targets):\n    if targets==1:\n        return \"insincere\"\n    else:\n        return \"sincere\"","metadata":{"execution":{"iopub.status.busy":"2024-03-11T04:16:55.850044Z","iopub.execute_input":"2024-03-11T04:16:55.850619Z","iopub.status.idle":"2024-03-11T04:16:55.855606Z","shell.execute_reply.started":"2024-03-11T04:16:55.850588Z","shell.execute_reply":"2024-03-11T04:16:55.854834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp_df['target'] = temp_df['target'].apply(ratings)\nplt.pie(df['target'].value_counts(), labels=temp_df['target'].unique().tolist(), autopct='%1.1f%%')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-11T04:17:02.457417Z","iopub.execute_input":"2024-03-11T04:17:02.457909Z","iopub.status.idle":"2024-03-11T04:17:02.91768Z","shell.execute_reply.started":"2024-03-11T04:17:02.45788Z","shell.execute_reply":"2024-03-11T04:17:02.916153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lenght = len(df['question_text'][1])\nprint(f'Length of a sample review: {lenght}')","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:36:47.739202Z","iopub.execute_input":"2024-03-10T16:36:47.740155Z","iopub.status.idle":"2024-03-10T16:36:47.747408Z","shell.execute_reply.started":"2024-03-10T16:36:47.740113Z","shell.execute_reply":"2024-03-10T16:36:47.746368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def word_count(review):\n    review_list = review.split()\n    return len(review_list)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:36:48.75736Z","iopub.execute_input":"2024-03-10T16:36:48.757767Z","iopub.status.idle":"2024-03-10T16:36:48.7639Z","shell.execute_reply.started":"2024-03-10T16:36:48.757737Z","shell.execute_reply":"2024-03-10T16:36:48.76258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Word_count'] = df['question_text'].apply(word_count)\ndf.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:36:49.763564Z","iopub.execute_input":"2024-03-10T16:36:49.764062Z","iopub.status.idle":"2024-03-10T16:36:49.816109Z","shell.execute_reply.started":"2024-03-10T16:36:49.764028Z","shell.execute_reply":"2024-03-10T16:36:49.814897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Length'] = df['question_text'].str.len()\ndf.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:36:50.624063Z","iopub.execute_input":"2024-03-10T16:36:50.624467Z","iopub.status.idle":"2024-03-10T16:36:50.653634Z","shell.execute_reply.started":"2024-03-10T16:36:50.624439Z","shell.execute_reply":"2024-03-10T16:36:50.652416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['mean_sent_length'] = df['question_text'].map(lambda rev: np.mean([len(sent) for sent in tokenize.sent_tokenize(rev)]))\ndf.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:36:51.608839Z","iopub.execute_input":"2024-03-10T16:36:51.609226Z","iopub.status.idle":"2024-03-10T16:36:53.032583Z","shell.execute_reply.started":"2024-03-10T16:36:51.609198Z","shell.execute_reply":"2024-03-10T16:36:53.031335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def visualize(col):\n    plt.figure(figsize=(15, 8))\n    \n    plt.subplot(1, 2, 1)\n    sns.boxplot(y=df[col], hue=df['target'])\n    plt.ylabel(col, labelpad=11.5)\n    plt.ylim(-1*df[col].max()/10, df[col].max(),5000)\n    \n    plt.subplot(1, 2, 2)\n    for target in df['target'].unique():\n        sns.kdeplot(df[df['target'] == target][col], label=target)\n    plt.xlabel('')\n    plt.ylabel('')\n    plt.legend(df['target'].unique())\n    plt.xlim(-1*df[col].max()/10, df[col].max())\n\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:51:50.134587Z","iopub.execute_input":"2024-03-10T16:51:50.134994Z","iopub.status.idle":"2024-03-10T16:51:50.147929Z","shell.execute_reply.started":"2024-03-10T16:51:50.134965Z","shell.execute_reply":"2024-03-10T16:51:50.146336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:50:33.783715Z","iopub.execute_input":"2024-03-10T16:50:33.784234Z","iopub.status.idle":"2024-03-10T16:50:33.790562Z","shell.execute_reply.started":"2024-03-10T16:50:33.784201Z","shell.execute_reply":"2024-03-10T16:50:33.789068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\ndef visualize(col):\n    plt.figure(figsize=(15, 8))\n\n    plt.subplot(1, 2, 1)\n    sns.boxplot(x='target', y=col, data=df)\n    plt.ylabel(col, labelpad=12.5)\n\n    plt.subplot(1, 2, 2)\n    sns.kdeplot(df[df['target'] == 0][col], label='sincere', shade=True)\n    sns.kdeplot(df[df['target'] == 1][col], label='insincere', shade=True)\n    plt.xlabel('')\n    plt.ylabel('')\n    plt.legend()\n\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:51:32.152608Z","iopub.execute_input":"2024-03-10T16:51:32.153031Z","iopub.status.idle":"2024-03-10T16:51:32.162133Z","shell.execute_reply.started":"2024-03-10T16:51:32.152998Z","shell.execute_reply":"2024-03-10T16:51:32.160991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = df.columns.tolist()[2:]\nfor feature in features:\n    visualize(feature)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:51:53.310594Z","iopub.execute_input":"2024-03-10T16:51:53.311041Z","iopub.status.idle":"2024-03-10T16:51:55.259429Z","shell.execute_reply.started":"2024-03-10T16:51:53.311008Z","shell.execute_reply":"2024-03-10T16:51:55.258219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#deleting duplicate rows\ndf.drop_duplicates(inplace=True)\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:54:13.387971Z","iopub.execute_input":"2024-03-10T16:54:13.388511Z","iopub.status.idle":"2024-03-10T16:54:13.412131Z","shell.execute_reply.started":"2024-03-10T16:54:13.388381Z","shell.execute_reply":"2024-03-10T16:54:13.410891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#converting to lower case\ndf['question_text'] = df['question_text'].str.lower()\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:54:21.717171Z","iopub.execute_input":"2024-03-10T16:54:21.717584Z","iopub.status.idle":"2024-03-10T16:54:21.742012Z","shell.execute_reply.started":"2024-03-10T16:54:21.717551Z","shell.execute_reply":"2024-03-10T16:54:21.740747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#remove urls\nimport re\ndef remove_url(text):\n    pattern = re.compile(r'https?://\\S+|www\\.\\S+')\n    return pattern.sub(r'', text)\n\ndf['question_text'] = df['question_text'].apply(remove_url)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:54:27.752893Z","iopub.execute_input":"2024-03-10T16:54:27.753809Z","iopub.status.idle":"2024-03-10T16:54:27.83367Z","shell.execute_reply.started":"2024-03-10T16:54:27.753773Z","shell.execute_reply":"2024-03-10T16:54:27.832617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['question_text'][3]","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:54:30.776024Z","iopub.execute_input":"2024-03-10T16:54:30.777009Z","iopub.status.idle":"2024-03-10T16:54:30.785542Z","shell.execute_reply.started":"2024-03-10T16:54:30.776964Z","shell.execute_reply":"2024-03-10T16:54:30.784082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#remove_html_tags\ndef remove_html_tags(text):\n    pattern = re.compile('<.*?>')\n    return pattern.sub(r'', text)\n\ndf['question_text'] = df['question_text'].apply(remove_html_tags)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:54:33.915392Z","iopub.execute_input":"2024-03-10T16:54:33.915797Z","iopub.status.idle":"2024-03-10T16:54:33.958978Z","shell.execute_reply.started":"2024-03-10T16:54:33.915767Z","shell.execute_reply":"2024-03-10T16:54:33.95776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#remove punctuation\ndef remove_punc1(text):\n    exclude = '''!()-[]{};:'\"\\,<>./?@#$%^&*_~'''\n    return text.translate(str.maketrans('', '', exclude))\n\ndf['question_text'] = df['question_text'].apply(remove_punc1)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:54:35.849218Z","iopub.execute_input":"2024-03-10T16:54:35.850337Z","iopub.status.idle":"2024-03-10T16:54:35.957565Z","shell.execute_reply.started":"2024-03-10T16:54:35.85029Z","shell.execute_reply":"2024-03-10T16:54:35.956157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['question_text'][0]","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:54:37.88293Z","iopub.execute_input":"2024-03-10T16:54:37.884159Z","iopub.status.idle":"2024-03-10T16:54:37.891788Z","shell.execute_reply.started":"2024-03-10T16:54:37.884111Z","shell.execute_reply":"2024-03-10T16:54:37.890741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Stopword Removal","metadata":{}},{"cell_type":"code","source":"from nltk.corpus import stopwords\n\nsw_list = stopwords.words('english')\ndf['question_text'] = df['question_text'].apply(lambda x: [item for item in x.split() if item not in sw_list])\n.apply(lambda x:\" \".join(x))","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:54:42.218236Z","iopub.execute_input":"2024-03-10T16:54:42.218716Z","iopub.status.idle":"2024-03-10T16:54:42.795771Z","shell.execute_reply.started":"2024-03-10T16:54:42.218677Z","shell.execute_reply":"2024-03-10T16:54:42.794638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['question_text'][0]\n","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:54:47.056364Z","iopub.execute_input":"2024-03-10T16:54:47.056805Z","iopub.status.idle":"2024-03-10T16:54:47.063988Z","shell.execute_reply.started":"2024-03-10T16:54:47.056773Z","shell.execute_reply":"2024-03-10T16:54:47.063069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def corpus(text):\n    text_list = text.split()\n    return text_list","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:55:18.526567Z","iopub.execute_input":"2024-03-10T16:55:18.527755Z","iopub.status.idle":"2024-03-10T16:55:18.532962Z","shell.execute_reply.started":"2024-03-10T16:55:18.527711Z","shell.execute_reply":"2024-03-10T16:55:18.531698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['question_lists'] = df['question_text'].apply(corpus)\ndf.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:56:24.295103Z","iopub.execute_input":"2024-03-10T16:56:24.295519Z","iopub.status.idle":"2024-03-10T16:56:24.344245Z","shell.execute_reply.started":"2024-03-10T16:56:24.295489Z","shell.execute_reply":"2024-03-10T16:56:24.342979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corpus = []\nfor i in trange(df.shape[0], ncols=150, nrows=10, colour='green', smoothing=0.8):\n    corpus += df['question_lists'][i]\nlen(corpus)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:56:47.584563Z","iopub.execute_input":"2024-03-10T16:56:47.584983Z","iopub.status.idle":"2024-03-10T16:56:47.852401Z","shell.execute_reply.started":"2024-03-10T16:56:47.584953Z","shell.execute_reply":"2024-03-10T16:56:47.85123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mostCommon = Counter(corpus).most_common(10)\nmostCommon","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:57:05.029364Z","iopub.execute_input":"2024-03-10T16:57:05.029822Z","iopub.status.idle":"2024-03-10T16:57:05.071897Z","shell.execute_reply.started":"2024-03-10T16:57:05.029788Z","shell.execute_reply":"2024-03-10T16:57:05.07081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"words = []\nfreq = []\nfor word, count in mostCommon:\n    words.append(word)\n    freq.append(count)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:57:32.88632Z","iopub.execute_input":"2024-03-10T16:57:32.886702Z","iopub.status.idle":"2024-03-10T16:57:32.893566Z","shell.execute_reply.started":"2024-03-10T16:57:32.886674Z","shell.execute_reply":"2024-03-10T16:57:32.89214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.barplot(x=freq, y=words)\nplt.title('Top 10 Most Frequently Occuring Words')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:57:50.012208Z","iopub.execute_input":"2024-03-10T16:57:50.01258Z","iopub.status.idle":"2024-03-10T16:57:50.288188Z","shell.execute_reply.started":"2024-03-10T16:57:50.012553Z","shell.execute_reply":"2024-03-10T16:57:50.28705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv = CountVectorizer(ngram_range=(2,2))\nbigrams = cv.fit_transform(df['question_text'])","metadata":{"execution":{"iopub.status.busy":"2024-03-10T17:00:04.304746Z","iopub.execute_input":"2024-03-10T17:00:04.305174Z","iopub.status.idle":"2024-03-10T17:00:05.075966Z","shell.execute_reply.started":"2024-03-10T17:00:04.305142Z","shell.execute_reply":"2024-03-10T17:00:05.074702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count_values = bigrams.toarray().sum(axis=0)\nngram_freq = pd.DataFrame(sorted([(count_values[i], k) for k, i in cv.vocabulary_.items()], reverse = True))\nngram_freq.columns = [\"frequency\", \"bigram\"]","metadata":{"execution":{"iopub.status.busy":"2024-03-10T17:03:22.844936Z","iopub.execute_input":"2024-03-10T17:03:22.84535Z","iopub.status.idle":"2024-03-10T17:03:29.938126Z","shell.execute_reply.started":"2024-03-10T17:03:22.845317Z","shell.execute_reply":"2024-03-10T17:03:29.937106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.barplot(x=ngram_freq['frequency'][:10], y=ngram_freq['bigram'][:10])\nplt.title('Top 10 Most Frequently Occuring Bigrams')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-10T17:03:29.940306Z","iopub.execute_input":"2024-03-10T17:03:29.940789Z","iopub.status.idle":"2024-03-10T17:03:30.222506Z","shell.execute_reply.started":"2024-03-10T17:03:29.940745Z","shell.execute_reply":"2024-03-10T17:03:30.221472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['question_text'][3]","metadata":{"execution":{"iopub.status.busy":"2024-03-10T09:52:09.53578Z","iopub.execute_input":"2024-03-10T09:52:09.536069Z","iopub.status.idle":"2024-03-10T09:52:09.542173Z","shell.execute_reply.started":"2024-03-10T09:52:09.536047Z","shell.execute_reply":"2024-03-10T09:52:09.54109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gensim\nfrom nltk import word_tokenize\nfrom gensim.utils import simple_preprocess\nlemmatizer = WordNetLemmatizer()\nfrom nltk.tokenize import word_tokenize\nfrom nltk.stem import WordNetLemmatizer","metadata":{"execution":{"iopub.status.busy":"2024-03-10T09:50:28.09638Z","iopub.execute_input":"2024-03-10T09:50:28.096701Z","iopub.status.idle":"2024-03-10T09:50:28.100301Z","shell.execute_reply.started":"2024-03-10T09:50:28.09668Z","shell.execute_reply":"2024-03-10T09:50:28.099731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tokenization and Lemmatization","metadata":{}},{"cell_type":"code","source":"def tokenize_and_lemmatize(text):\n    tokens = word_tokenize(text)\n    lemmatized_tokens = [lemmatizer.lemmatize(token) for token in tokens]\n    lemmatized_string = ' '.join(lemmatized_tokens) \n    return lemmatized_string","metadata":{"execution":{"iopub.status.busy":"2024-03-10T09:51:33.581997Z","iopub.execute_input":"2024-03-10T09:51:33.582288Z","iopub.status.idle":"2024-03-10T09:51:33.587471Z","shell.execute_reply.started":"2024-03-10T09:51:33.582267Z","shell.execute_reply":"2024-03-10T09:51:33.586661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['lematize'] = df['question_text'].apply(tokenize_and_lemmatize)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T09:54:14.996083Z","iopub.execute_input":"2024-03-10T09:54:14.99638Z","iopub.status.idle":"2024-03-10T09:54:17.392863Z","shell.execute_reply.started":"2024-03-10T09:54:14.996358Z","shell.execute_reply":"2024-03-10T09:54:17.392033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lemmatizer.lemmatize('crying')\ndf","metadata":{"execution":{"iopub.status.busy":"2024-03-10T09:54:33.974715Z","iopub.execute_input":"2024-03-10T09:54:33.975001Z","iopub.status.idle":"2024-03-10T09:54:33.98502Z","shell.execute_reply.started":"2024-03-10T09:54:33.97498Z","shell.execute_reply":"2024-03-10T09:54:33.984304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import nltk\nimport subprocess\n\n# Download and unzip wordnet\ntry:\n    nltk.data.find('wordnet.zip')\nexcept:\n    nltk.download('wordnet', download_dir='/kaggle/working/')\n    command = \"unzip /kaggle/working/corpora/wordnet.zip -d /kaggle/working/corpora\"\n    subprocess.run(command.split())\n    nltk.data.path.append('/kaggle/working/')\n\n# Now you can import the NLTK resources as usual\nfrom nltk.corpus import wordnet","metadata":{"execution":{"iopub.status.busy":"2024-03-10T09:16:29.62529Z","iopub.execute_input":"2024-03-10T09:16:29.62565Z","iopub.status.idle":"2024-03-10T09:16:29.921015Z","shell.execute_reply.started":"2024-03-10T09:16:29.625626Z","shell.execute_reply":"2024-03-10T09:16:29.920337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom nltk.corpus import wordnet\nfrom nltk.stem import WordNetLemmatizer\nlemmatizer = nltk.stem.WordNetLemmatizer()\nlem=[]\nfor i in story:\n    for j in i:\n        lem.append(lemmatizer.lemmatize(j))","metadata":{"execution":{"iopub.status.busy":"2024-03-10T09:25:20.644843Z","iopub.execute_input":"2024-03-10T09:25:20.645151Z","iopub.status.idle":"2024-03-10T09:25:21.261351Z","shell.execute_reply.started":"2024-03-10T09:25:20.645129Z","shell.execute_reply":"2024-03-10T09:25:21.260716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"comp=[]\nfor i in story:\n    for j in i:\n        comp.append(j)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T09:25:27.815155Z","iopub.execute_input":"2024-03-10T09:25:27.815439Z","iopub.status.idle":"2024-03-10T09:25:27.855358Z","shell.execute_reply.started":"2024-03-10T09:25:27.815416Z","shell.execute_reply":"2024-03-10T09:25:27.854522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(comp)):\n    if(comp[i] != lem[i]):\n        print(comp[i],\"==>\",lem[i])","metadata":{"execution":{"iopub.status.busy":"2024-03-10T09:27:12.917281Z","iopub.execute_input":"2024-03-10T09:27:12.917642Z","iopub.status.idle":"2024-03-10T09:27:13.096055Z","shell.execute_reply.started":"2024-03-10T09:27:12.917616Z","shell.execute_reply":"2024-03-10T09:27:13.095433Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lem","metadata":{"execution":{"iopub.status.busy":"2024-03-10T09:22:45.656942Z","iopub.execute_input":"2024-03-10T09:22:45.657249Z","iopub.status.idle":"2024-03-10T09:22:45.674761Z","shell.execute_reply.started":"2024-03-10T09:22:45.657227Z","shell.execute_reply":"2024-03-10T09:22:45.673918Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --upgrade nltk\n","metadata":{"execution":{"iopub.status.busy":"2024-03-10T08:53:22.130259Z","iopub.execute_input":"2024-03-10T08:53:22.130579Z","iopub.status.idle":"2024-03-10T08:53:32.849741Z","shell.execute_reply.started":"2024-03-10T08:53:22.130556Z","shell.execute_reply":"2024-03-10T08:53:32.848821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install cmake","metadata":{"execution":{"iopub.status.busy":"2024-03-10T08:48:00.534061Z","iopub.execute_input":"2024-03-10T08:48:00.534346Z","iopub.status.idle":"2024-03-10T08:48:12.606855Z","shell.execute_reply.started":"2024-03-10T08:48:00.534324Z","shell.execute_reply":"2024-03-10T08:48:12.605922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mkdir ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model=gensim.models.Word2Vec(\nwindow=10,\nmin_count=2)","metadata":{"execution":{"iopub.status.busy":"2024-02-08T10:55:16.682231Z","iopub.execute_input":"2024-02-08T10:55:16.682617Z","iopub.status.idle":"2024-02-08T10:55:16.69553Z","shell.execute_reply.started":"2024-02-08T10:55:16.682589Z","shell.execute_reply":"2024-02-08T10:55:16.693868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.build_vocab(story)","metadata":{"execution":{"iopub.status.busy":"2024-02-08T10:55:19.347639Z","iopub.execute_input":"2024-02-08T10:55:19.348079Z","iopub.status.idle":"2024-02-08T10:55:22.281469Z","shell.execute_reply.started":"2024-02-08T10:55:19.348045Z","shell.execute_reply":"2024-02-08T10:55:22.280178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.train(story,total_examples=model.corpus_count,epochs=model.epochs)","metadata":{"execution":{"iopub.status.busy":"2024-02-08T10:55:27.48617Z","iopub.execute_input":"2024-02-08T10:55:27.486586Z","iopub.status.idle":"2024-02-08T10:55:45.942975Z","shell.execute_reply.started":"2024-02-08T10:55:27.486542Z","shell.execute_reply":"2024-02-08T10:55:45.941619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(model.wv.index_to_key)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-08T10:55:51.456892Z","iopub.execute_input":"2024-02-08T10:55:51.457321Z","iopub.status.idle":"2024-02-08T10:55:51.465182Z","shell.execute_reply.started":"2024-02-08T10:55:51.457291Z","shell.execute_reply":"2024-02-08T10:55:51.464009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def document_vector(doc):\n    doc = [word for word in doc.split() if word in model.wv.index_to_key]\n    return np.mean(model.wv[doc], axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-02-08T10:55:55.150241Z","iopub.execute_input":"2024-02-08T10:55:55.150636Z","iopub.status.idle":"2024-02-08T10:55:55.156421Z","shell.execute_reply.started":"2024-02-08T10:55:55.150607Z","shell.execute_reply":"2024-02-08T10:55:55.155422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"document_vector(df['question_text'].values[0])","metadata":{"execution":{"iopub.status.busy":"2024-02-08T10:55:58.937458Z","iopub.execute_input":"2024-02-08T10:55:58.937981Z","iopub.status.idle":"2024-02-08T10:55:58.956668Z","shell.execute_reply.started":"2024-02-08T10:55:58.937947Z","shell.execute_reply":"2024-02-08T10:55:58.95531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2024-02-08T10:56:01.452919Z","iopub.execute_input":"2024-02-08T10:56:01.453324Z","iopub.status.idle":"2024-02-08T10:56:01.458968Z","shell.execute_reply.started":"2024-02-08T10:56:01.453292Z","shell.execute_reply":"2024-02-08T10:56:01.457505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X=[]\nfor doc in tqdm(df['question_text'].values):\n    vector=(document_vector(doc))\n    vector\n    if vector is not None and len(vector) > 0:\n        X.append(vector)","metadata":{"execution":{"iopub.status.busy":"2024-02-08T10:58:49.927863Z","iopub.execute_input":"2024-02-08T10:58:49.928687Z","iopub.status.idle":"2024-02-08T10:58:50.028466Z","shell.execute_reply.started":"2024-02-08T10:58:49.928651Z","shell.execute_reply":"2024-02-08T10:58:50.026514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}