{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-11T06:33:03.552647Z","iopub.execute_input":"2022-08-11T06:33:03.553486Z","iopub.status.idle":"2022-08-11T06:33:03.589754Z","shell.execute_reply.started":"2022-08-11T06:33:03.553387Z","shell.execute_reply":"2022-08-11T06:33:03.588197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import nltk\nnltk.download('omw-1.4')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:43:01.766158Z","iopub.execute_input":"2022-08-11T06:43:01.766630Z","iopub.status.idle":"2022-08-11T06:43:02.065004Z","shell.execute_reply.started":"2022-08-11T06:43:01.766594Z","shell.execute_reply":"2022-08-11T06:43:02.063749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Hello world\nI have clearly explained the preprocessing steps in a previous notebook\nHere is the link if you have some trouble following the code for data processing https://www.kaggle.com/code/ajayuser/nlp-with-disaster-tweets-sklearn\n\nIf you like to see image preprocessing techniques in python, go checkout https://www.kaggle.com/code/ajayuser/facial-key-points-detection-preprocessing-guide\n\nToday the main goal is to train a TENSORFLOW MODEL\n\nWe will build a simple Bi-directional RNN for text classification\n\nIf you like to see a logistic regression model checkout : https://www.kaggle.com/code/ajayuser/nlp-classif-tensorflow-logistic-regression\n\nAs always please post your suggestions, reviews, recommendations in the comments section","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:35:10.010274Z","iopub.execute_input":"2022-08-11T06:35:10.010750Z","iopub.status.idle":"2022-08-11T06:35:10.019948Z","shell.execute_reply.started":"2022-08-11T06:35:10.010716Z","shell.execute_reply":"2022-08-11T06:35:10.018231Z"}}},{"cell_type":"code","source":"import re\nimport string\nimport tensorflow as tf\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize\nfrom nltk.stem import PorterStemmer\nfrom nltk.stem import WordNetLemmatizer\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:56:28.318811Z","iopub.execute_input":"2022-08-11T06:56:28.319201Z","iopub.status.idle":"2022-08-11T06:56:28.506051Z","shell.execute_reply.started":"2022-08-11T06:56:28.319170Z","shell.execute_reply":"2022-08-11T06:56:28.504813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/nlp-getting-started/train.csv')\ntest_df = pd.read_csv('/kaggle/input/nlp-getting-started/test.csv')\nsample_submission_df = pd.read_csv('/kaggle/input/nlp-getting-started/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:36:01.997300Z","iopub.execute_input":"2022-08-11T06:36:01.998079Z","iopub.status.idle":"2022-08-11T06:36:02.087531Z","shell.execute_reply.started":"2022-08-11T06:36:01.998027Z","shell.execute_reply":"2022-08-11T06:36:02.085909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:36:30.316624Z","iopub.execute_input":"2022-08-11T06:36:30.317516Z","iopub.status.idle":"2022-08-11T06:36:30.343809Z","shell.execute_reply.started":"2022-08-11T06:36:30.317465Z","shell.execute_reply":"2022-08-11T06:36:30.342865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data preprocessing","metadata":{}},{"cell_type":"code","source":"from sklearn.base import BaseEstimator, TransformerMixin\n\nclass textTransformer(TransformerMixin):\n    def __init__(self):\n        self.lemmatizer = WordNetLemmatizer()\n        self.stemmer = PorterStemmer()\n        self.english_stop_words = stopwords.words('english')\n    \n    def remove_html(self, text):\n        return re.sub(pattern=r'<\\w+/?>', repl='',string=text)\n    \n    def remove_mentioning(self, text):\n        return re.sub(pattern=r'@[^\\s]*', repl='',string=text)\n\n    def remove_url(self, text):\n        return re.sub(pattern=r'http[^\\s]*', repl='', string=text)\n    \n    def remove_punctuation(self, text):\n        return re.sub(pattern=r\"[{0}]\".format(string.punctuation),repl='',string=text)\n    \n    def remove_numbers(self, text):\n        return re.sub(pattern=r'\\d+', repl='', string=text)\n    \n    def remove_stop_words(self, text):\n        return ' '.join([word for word in word_tokenize(text) if word not in self.english_stop_words])\n    \n    def stemming(self, text):\n        return ' '.join([self.stemmer.stem(word) for word in word_tokenize(text)])\n    \n    def lemmatize(self, text, pos='v'):\n        return ' '.join([self.lemmatizer.lemmatize(word, pos=pos) for word in word_tokenize(text)])\n    \n    def fit(self,X):\n        return self\n    \n    def transform(self,X):\n        \n\n        X['processed_text']= (X['text']\n                              .apply(lambda text: text.lower())\n                              .apply(self.remove_html)\n                              .apply(self.remove_url)\n                              .apply(self.remove_mentioning)\n                              .apply(self.remove_punctuation)\n                              .apply(self.remove_numbers)\n                              .apply(self.remove_stop_words)\n                             )\n                           \n    \n            \n        X['lemmatized_text']= X['processed_text'].apply(self.lemmatize)\n        \n        X['stemmed_text']= X['processed_text'].apply(self.stemming)\n\n        \n        return X\n        ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:43:06.924796Z","iopub.execute_input":"2022-08-11T06:43:06.925216Z","iopub.status.idle":"2022-08-11T06:43:06.941952Z","shell.execute_reply.started":"2022-08-11T06:43:06.925180Z","shell.execute_reply":"2022-08-11T06:43:06.940326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Process Training data","metadata":{}},{"cell_type":"code","source":"%time\n# APPLY THE TRANSFORMATION\ntransformer = textTransformer()\n\nX = transformer.fit_transform(train_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:43:09.747820Z","iopub.execute_input":"2022-08-11T06:43:09.748222Z","iopub.status.idle":"2022-08-11T06:43:17.447507Z","shell.execute_reply.started":"2022-08-11T06:43:09.748189Z","shell.execute_reply":"2022-08-11T06:43:17.446160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CHECK THE DATA\nX[['text','processed_text','lemmatized_text','stemmed_text']]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:43:25.002336Z","iopub.execute_input":"2022-08-11T06:43:25.002783Z","iopub.status.idle":"2022-08-11T06:43:25.026553Z","shell.execute_reply.started":"2022-08-11T06:43:25.002736Z","shell.execute_reply":"2022-08-11T06:43:25.025638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# APPLY THE TRANSFORMATION TO TEST DATA\ntransformer = textTransformer()\n\nX_test = transformer.transform(test_df)\n\nX_test[['text','processed_text','lemmatized_text','stemmed_text']]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:45:18.854923Z","iopub.execute_input":"2022-08-11T06:45:18.855530Z","iopub.status.idle":"2022-08-11T06:45:21.304287Z","shell.execute_reply.started":"2022-08-11T06:45:18.855470Z","shell.execute_reply":"2022-08-11T06:45:21.302836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create Dataset","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:44:43.877919Z","iopub.execute_input":"2022-08-11T06:44:43.878408Z","iopub.status.idle":"2022-08-11T06:44:43.883607Z","shell.execute_reply.started":"2022-08-11T06:44:43.878366Z","shell.execute_reply":"2022-08-11T06:44:43.882328Z"}}},{"cell_type":"code","source":"# Create a dataset\n\ntrain_ds = tf.data.Dataset.from_tensor_slices((X['lemmatized_text'],train_df['target']))\ntest_ds = tf.data.Dataset.from_tensor_slices((X_test['lemmatized_text']))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:45:27.960391Z","iopub.execute_input":"2022-08-11T06:45:27.960940Z","iopub.status.idle":"2022-08-11T06:45:28.024094Z","shell.execute_reply.started":"2022-08-11T06:45:27.960899Z","shell.execute_reply":"2022-08-11T06:45:28.022860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds = train_ds.batch(64).prefetch(tf.data.AUTOTUNE)\ntest_ds = test_ds.batch(64).prefetch(tf.data.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:45:42.116946Z","iopub.execute_input":"2022-08-11T06:45:42.117423Z","iopub.status.idle":"2022-08-11T06:45:42.128365Z","shell.execute_reply.started":"2022-08-11T06:45:42.117389Z","shell.execute_reply":"2022-08-11T06:45:42.127037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_batch, label_batch = next(iter(train_ds))\n\nfor text, label in zip(text_batch,label_batch):\n    print(text)\n    print('label',label)\n    break","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:45:54.070375Z","iopub.execute_input":"2022-08-11T06:45:54.070883Z","iopub.status.idle":"2022-08-11T06:45:54.141477Z","shell.execute_reply.started":"2022-08-11T06:45:54.070833Z","shell.execute_reply":"2022-08-11T06:45:54.139958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Vocabulary","metadata":{}},{"cell_type":"code","source":"# lets find out the sequence length of different rows\n# create a histogram using the seq-len\n# and decide the maximum token-length \n# X[['text','processed_text','lemmatized_text','stemmed_text']]\nseq_len = []\nfor text in X['lemmatized_text']:\n    seq_len.append(len(word_tokenize(text)))\n\n\nplt.figure(figsize=(10,6))\nplt.subplot(121)\nplt.hist(seq_len);\nplt.subplot(122)\nsns.histplot(seq_len, kde=True);","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:57:14.906814Z","iopub.execute_input":"2022-08-11T06:57:14.907240Z","iopub.status.idle":"2022-08-11T06:57:16.235651Z","shell.execute_reply.started":"2022-08-11T06:57:14.907206Z","shell.execute_reply":"2022-08-11T06:57:16.234386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAX_TOKENS = 10000\nMAX_SEQ_LEN = 16","metadata":{"execution":{"iopub.status.busy":"2022-08-11T07:03:33.334452Z","iopub.execute_input":"2022-08-11T07:03:33.334873Z","iopub.status.idle":"2022-08-11T07:03:33.339763Z","shell.execute_reply.started":"2022-08-11T07:03:33.334840Z","shell.execute_reply":"2022-08-11T07:03:33.338582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# STANDARDIZE TOKENIZE VECTORIZE\ntext_vectorizer = tf.keras.layers.TextVectorization(\n    max_tokens=MAX_TOKENS,\n    output_mode='int',\n    output_sequence_length=MAX_SEQ_LEN\n)\n\n\n# ADAPT\n\ntrain_text = train_ds.map(lambda x,y:x, num_parallel_calls=tf.data.AUTOTUNE)\n\ntext_vectorizer.adapt(train_text)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T07:03:35.669712Z","iopub.execute_input":"2022-08-11T07:03:35.670104Z","iopub.status.idle":"2022-08-11T07:03:35.870304Z","shell.execute_reply.started":"2022-08-11T07:03:35.670066Z","shell.execute_reply":"2022-08-11T07:03:35.868850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lets check our vocab\ntext_vectorizer.get_vocabulary()[:10]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T07:03:40.296391Z","iopub.execute_input":"2022-08-11T07:03:40.296898Z","iopub.status.idle":"2022-08-11T07:03:40.332065Z","shell.execute_reply.started":"2022-08-11T07:03:40.296854Z","shell.execute_reply":"2022-08-11T07:03:40.330443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modeling ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T07:01:10.026628Z","iopub.execute_input":"2022-08-11T07:01:10.027847Z","iopub.status.idle":"2022-08-11T07:01:10.098895Z","shell.execute_reply.started":"2022-08-11T07:01:10.027795Z","shell.execute_reply":"2022-08-11T07:01:10.097526Z"}}},{"cell_type":"code","source":"DIMS = 64","metadata":{"execution":{"iopub.status.busy":"2022-08-11T07:08:14.254837Z","iopub.execute_input":"2022-08-11T07:08:14.255329Z","iopub.status.idle":"2022-08-11T07:08:14.261487Z","shell.execute_reply.started":"2022-08-11T07:08:14.255290Z","shell.execute_reply":"2022-08-11T07:08:14.259946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = tf.keras.Sequential([\n    text_vectorizer,     # [batch, seq]\n    tf.keras.layers.Embedding(MAX_TOKENS, DIMS, mask_zero=True),  # [batch, seq, dims]\n    tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(DIMS, return_sequences=True)), # [batch, seq, dims]\n    tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(DIMS)), # [batch, dims]\n    tf.keras.layers.Dropout(0.1),\n#     tf.keras.layers.Dense(32,'relu'),\n    tf.keras.layers.Dense(1)\n])\n\n\nmodel.compile(loss=tf.keras.losses.BinaryCrossentropy(from_logits=True),\n              optimizer=tf.keras.optimizers.Adam(1e-4),\n              metrics=['accuracy'])\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T07:36:16.220012Z","iopub.execute_input":"2022-08-11T07:36:16.220397Z","iopub.status.idle":"2022-08-11T07:36:19.631535Z","shell.execute_reply.started":"2022-08-11T07:36:16.220367Z","shell.execute_reply":"2022-08-11T07:36:19.630295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train_ds, epochs=10)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T07:36:22.556616Z","iopub.execute_input":"2022-08-11T07:36:22.557033Z","iopub.status.idle":"2022-08-11T07:38:03.400469Z","shell.execute_reply.started":"2022-08-11T07:36:22.556998Z","shell.execute_reply":"2022-08-11T07:38:03.399430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prediction and submission","metadata":{}},{"cell_type":"code","source":"preds = model.predict(test_ds)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T07:38:08.984058Z","iopub.execute_input":"2022-08-11T07:38:08.984492Z","iopub.status.idle":"2022-08-11T07:38:14.637766Z","shell.execute_reply.started":"2022-08-11T07:38:08.984457Z","shell.execute_reply":"2022-08-11T07:38:14.636841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"threshold = 0.2\n\nprediction_df = pd.DataFrame(data={'text':test_df['text'],'label':np.where(preds<threshold,0,1).ravel()})\n\nprediction_df.head(25)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T07:42:23.261549Z","iopub.execute_input":"2022-08-11T07:42:23.261994Z","iopub.status.idle":"2022-08-11T07:42:23.277670Z","shell.execute_reply.started":"2022-08-11T07:42:23.261960Z","shell.execute_reply":"2022-08-11T07:42:23.276652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"sample_submission_df['target']=np.where(preds<threshold,0,1).ravel()\nsample_submission_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T07:38:53.970184Z","iopub.execute_input":"2022-08-11T07:38:53.970589Z","iopub.status.idle":"2022-08-11T07:38:53.982516Z","shell.execute_reply.started":"2022-08-11T07:38:53.970556Z","shell.execute_reply":"2022-08-11T07:38:53.981344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission_df.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T07:38:58.216867Z","iopub.execute_input":"2022-08-11T07:38:58.217653Z","iopub.status.idle":"2022-08-11T07:38:58.229699Z","shell.execute_reply.started":"2022-08-11T07:38:58.217601Z","shell.execute_reply":"2022-08-11T07:38:58.228663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}