{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-11T06:19:23.614790Z","iopub.execute_input":"2022-08-11T06:19:23.615674Z","iopub.status.idle":"2022-08-11T06:19:23.649169Z","shell.execute_reply.started":"2022-08-11T06:19:23.615550Z","shell.execute_reply":"2022-08-11T06:19:23.648242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Hello world\n\nI have clearly explained the preprocessing steps in a previous notebook<br>\nHere is the link if you have some trouble following the code for data processing\nhttps://www.kaggle.com/code/ajayuser/nlp-with-disaster-tweets-sklearn\n\nIf you like to see image preprocessing techniques in python, go checkout https://www.kaggle.com/code/ajayuser/facial-key-points-detection-preprocessing-guide\n\n\nToday the main goal is to train a TENSORFLOW MODEL\n\nWe will build a simple logistic regression classifier\n\nAs always please post your suggestions, reviews, recommendations in the comments section","metadata":{}},{"cell_type":"code","source":"import re\nimport string\nimport tensorflow as tf\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:19:23.651060Z","iopub.execute_input":"2022-08-11T06:19:23.651314Z","iopub.status.idle":"2022-08-11T06:19:31.021460Z","shell.execute_reply.started":"2022-08-11T06:19:23.651282Z","shell.execute_reply":"2022-08-11T06:19:31.020722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/nlp-getting-started/train.csv')\ntest_df = pd.read_csv('/kaggle/input/nlp-getting-started/test.csv')\nsample_submission_df = pd.read_csv('/kaggle/input/nlp-getting-started/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:19:31.022725Z","iopub.execute_input":"2022-08-11T06:19:31.023080Z","iopub.status.idle":"2022-08-11T06:19:31.117205Z","shell.execute_reply.started":"2022-08-11T06:19:31.023046Z","shell.execute_reply":"2022-08-11T06:19:31.115985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:19:31.120258Z","iopub.execute_input":"2022-08-11T06:19:31.120595Z","iopub.status.idle":"2022-08-11T06:19:31.143822Z","shell.execute_reply.started":"2022-08-11T06:19:31.120553Z","shell.execute_reply":"2022-08-11T06:19:31.142918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing","metadata":{}},{"cell_type":"code","source":"#  REMOVE <html> tags\nexample = '<p> hello <br> world <p/>'\nsample = re.sub(r'<[^><]*>','',example)\nprint(f'text before removing html tags : {example}\\ntext after removing html tags : {sample}')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:19:31.145763Z","iopub.execute_input":"2022-08-11T06:19:31.146392Z","iopub.status.idle":"2022-08-11T06:19:31.153646Z","shell.execute_reply.started":"2022-08-11T06:19:31.146332Z","shell.execute_reply":"2022-08-11T06:19:31.152485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# REMOVE urls\n\nexample = 'Free Ebay Sniping RT? http://t.co/B231Ul1O1K  get yours now'\n\nre.sub(pattern=r'http[^\\s]*', repl='', string=example)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:19:31.155258Z","iopub.execute_input":"2022-08-11T06:19:31.156190Z","iopub.status.idle":"2022-08-11T06:19:31.165916Z","shell.execute_reply.started":"2022-08-11T06:19:31.156139Z","shell.execute_reply":"2022-08-11T06:19:31.165041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  REMOVE numbers tags\nexample = 'Mr John Smith 132, My Street,Kingston, New York 12401 United States'\nsample = re.sub(r'\\d+','',example)\nprint(f'text before removing numbers : {example}\\ntext after removing numbers : {sample}')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:19:31.167394Z","iopub.execute_input":"2022-08-11T06:19:31.168064Z","iopub.status.idle":"2022-08-11T06:19:31.179273Z","shell.execute_reply.started":"2022-08-11T06:19:31.168018Z","shell.execute_reply":"2022-08-11T06:19:31.178380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# REMOVE mentioning\n\nexample = '@BritishBakeOff This has opened up a new level of reality show'\n\nre.sub(pattern=r'@[^\\s]*', repl='',string=example)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:19:31.180597Z","iopub.execute_input":"2022-08-11T06:19:31.181571Z","iopub.status.idle":"2022-08-11T06:19:31.194501Z","shell.execute_reply.started":"2022-08-11T06:19:31.181516Z","shell.execute_reply":"2022-08-11T06:19:31.193615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  REMOVE punctuations\nexample = 'Healthcare is a priority, #obamacare #us-election @biden @obama'\nsample = re.sub(r'[{0}]'.format(string.punctuation),'',example)\nprint(f'text before removing punctuations : {example}\\ntext after removing punctuations : {sample}')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:19:31.195888Z","iopub.execute_input":"2022-08-11T06:19:31.196677Z","iopub.status.idle":"2022-08-11T06:19:31.221183Z","shell.execute_reply.started":"2022-08-11T06:19:31.196612Z","shell.execute_reply":"2022-08-11T06:19:31.219985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# REMOVE stopwords\nfrom nltk.corpus import stopwords\n\nenglish_stop_words = stopwords.words('english')\n\nexample = 'US inflation eases in July as petrol prices drop'\n\n' '.join([word for word in example.split() if word not in english_stop_words])","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:19:31.224876Z","iopub.execute_input":"2022-08-11T06:19:31.225330Z","iopub.status.idle":"2022-08-11T06:19:31.250499Z","shell.execute_reply.started":"2022-08-11T06:19:31.225276Z","shell.execute_reply":"2022-08-11T06:19:31.249679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  Lemmatization with NLTK\n\nfrom nltk.stem import WordNetLemmatizer\n\nlemmatizer = WordNetLemmatizer()\n\nexample = 'Ferrari and redbull are the fastest cars on the track'\nsample = ' '.join([lemmatizer.lemmatize(word,pos='v') for word in example.split()])\nprint(f'text before lemmatization : {example}\\ntext after lemmatization : {sample}')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:19:31.252087Z","iopub.execute_input":"2022-08-11T06:19:31.252790Z","iopub.status.idle":"2022-08-11T06:19:33.332016Z","shell.execute_reply.started":"2022-08-11T06:19:31.252751Z","shell.execute_reply":"2022-08-11T06:19:33.331041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Stemming words with NLTK\n\nfrom nltk.stem import PorterStemmer\nstemmer = PorterStemmer()\n\nexample = \"Ferrari's new design is interesting \"\nsample = ' '.join([stemmer.stem(word) for word in example.split()])\nprint(f'text before Stemming : {example}\\ntext after Stemming : {sample}')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:19:33.333519Z","iopub.execute_input":"2022-08-11T06:19:33.333932Z","iopub.status.idle":"2022-08-11T06:19:33.341660Z","shell.execute_reply.started":"2022-08-11T06:19:33.333900Z","shell.execute_reply":"2022-08-11T06:19:33.339762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.base import BaseEstimator, TransformerMixin\n\nclass textTransformer(TransformerMixin):\n    def __init__(self):\n        self.lemmatizer = WordNetLemmatizer()\n        self.stemmer = PorterStemmer()\n        self.english_stop_words = stopwords.words('english')\n    \n    def remove_html(self, text):\n        return re.sub(pattern=r'<\\w+/?>', repl='',string=text)\n    \n    def remove_mentioning(self, text):\n        return re.sub(pattern=r'@[^\\s]*', repl='',string=text)\n\n    def remove_url(self, text):\n        return re.sub(pattern=r'http[^\\s]*', repl='', string=text)\n    \n    def remove_punctuation(self, text):\n        return re.sub(pattern=r\"[{0}]\".format(string.punctuation),repl='',string=text)\n    \n    def remove_numbers(self, text):\n        return re.sub(pattern=r'\\d+', repl='', string=text)\n    \n    def remove_stop_words(self, text):\n        return ' '.join([word for word in word_tokenize(text) if word not in english_stop_words])\n    \n    def stemming(self, text):\n        return ' '.join([self.stemmer.stem(word) for word in word_tokenize(text)])\n    \n    def lemmatize(self, text, pos='v'):\n        return ' '.join([self.lemmatizer.lemmatize(word, pos=pos) for word in word_tokenize(text)])\n    \n    def fit(self,X):\n        return self\n    \n    def transform(self,X):\n        \n\n        X['processed_text']= (X['text']\n                              .apply(lambda text: text.lower())\n                              .apply(self.remove_html)\n                              .apply(self.remove_url)\n                              .apply(self.remove_mentioning)\n                              .apply(self.remove_punctuation)\n                              .apply(self.remove_numbers)\n                              .apply(self.remove_stop_words)\n                             )\n                           \n    \n            \n        X['lemmatized_text']= X['processed_text'].apply(self.lemmatize)\n        \n        X['stemmed_text']= X['processed_text'].apply(self.stemming)\n\n        \n        return X\n        ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:19:33.343273Z","iopub.execute_input":"2022-08-11T06:19:33.343665Z","iopub.status.idle":"2022-08-11T06:19:33.361094Z","shell.execute_reply.started":"2022-08-11T06:19:33.343602Z","shell.execute_reply":"2022-08-11T06:19:33.360235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%time\n# APPLY THE TRANSFORMATION\ntransformer = textTransformer()\n\nX = transformer.fit_transform(train_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:19:33.362469Z","iopub.execute_input":"2022-08-11T06:19:33.363528Z","iopub.status.idle":"2022-08-11T06:19:38.924892Z","shell.execute_reply.started":"2022-08-11T06:19:33.363480Z","shell.execute_reply":"2022-08-11T06:19:38.923903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CHECK THE DATA\nX[['text','processed_text','lemmatized_text','stemmed_text']]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:19:38.926300Z","iopub.execute_input":"2022-08-11T06:19:38.926558Z","iopub.status.idle":"2022-08-11T06:19:38.957312Z","shell.execute_reply.started":"2022-08-11T06:19:38.926528Z","shell.execute_reply":"2022-08-11T06:19:38.956461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# APPLY THE TRANSFORMATION TO TEST DATA\ntransformer = textTransformer()\n\nX_test = transformer.transform(test_df)\n\nX_test[['text','processed_text','lemmatized_text','stemmed_text']]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:19:38.958641Z","iopub.execute_input":"2022-08-11T06:19:38.959277Z","iopub.status.idle":"2022-08-11T06:19:41.298411Z","shell.execute_reply.started":"2022-08-11T06:19:38.959230Z","shell.execute_reply":"2022-08-11T06:19:41.297584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a dataset\n\ntrain_ds = tf.data.Dataset.from_tensor_slices((X['lemmatized_text'],train_df['target']))\ntest_ds = tf.data.Dataset.from_tensor_slices((X_test['lemmatized_text']))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:19:41.299664Z","iopub.execute_input":"2022-08-11T06:19:41.299879Z","iopub.status.idle":"2022-08-11T06:19:41.352147Z","shell.execute_reply.started":"2022-08-11T06:19:41.299852Z","shell.execute_reply":"2022-08-11T06:19:41.351413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds = train_ds.batch(64).prefetch(tf.data.AUTOTUNE)\ntest_ds = test_ds.batch(64).prefetch(tf.data.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:19:41.353380Z","iopub.execute_input":"2022-08-11T06:19:41.353829Z","iopub.status.idle":"2022-08-11T06:19:41.361533Z","shell.execute_reply.started":"2022-08-11T06:19:41.353796Z","shell.execute_reply":"2022-08-11T06:19:41.360685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_batch, label_batch = next(iter(train_ds))\n\nfor text, label in zip(text_batch,label_batch):\n    print(text)\n    print('label',label)\n    break","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:19:41.362673Z","iopub.execute_input":"2022-08-11T06:19:41.363345Z","iopub.status.idle":"2022-08-11T06:19:41.430270Z","shell.execute_reply.started":"2022-08-11T06:19:41.363302Z","shell.execute_reply":"2022-08-11T06:19:41.429538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# STANDARDIZE TOKENIZE VECTORIZE\nbinary_vectorizer = tf.keras.layers.TextVectorization(\n    max_tokens=10000,\n    output_mode='multi_hot',\n)\n\n\n# ADAPT\n\ntrain_text = train_ds.map(lambda x,y:x, num_parallel_calls=tf.data.AUTOTUNE)\n\nbinary_vectorizer.adapt(train_text)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:19:41.431533Z","iopub.execute_input":"2022-08-11T06:19:41.431978Z","iopub.status.idle":"2022-08-11T06:19:43.026380Z","shell.execute_reply.started":"2022-08-11T06:19:41.431944Z","shell.execute_reply":"2022-08-11T06:19:43.025654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modeling","metadata":{}},{"cell_type":"code","source":"model = tf.keras.Sequential([\n    binary_vectorizer,\n    tf.keras.layers.Dense(1)\n])\n\nmodel.compile(optimizer='adam',\n              loss=tf.keras.losses.BinaryCrossentropy(from_logits=True),\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:19:43.027867Z","iopub.execute_input":"2022-08-11T06:19:43.028290Z","iopub.status.idle":"2022-08-11T06:19:43.348151Z","shell.execute_reply.started":"2022-08-11T06:19:43.028258Z","shell.execute_reply":"2022-08-11T06:19:43.347033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train_ds,epochs=25)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:19:43.349581Z","iopub.execute_input":"2022-08-11T06:19:43.349870Z","iopub.status.idle":"2022-08-11T06:19:55.361452Z","shell.execute_reply.started":"2022-08-11T06:19:43.349838Z","shell.execute_reply":"2022-08-11T06:19:55.360828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prediction and submission","metadata":{}},{"cell_type":"code","source":"preds = model.predict(test_ds)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:19:55.363693Z","iopub.execute_input":"2022-08-11T06:19:55.363936Z","iopub.status.idle":"2022-08-11T06:19:55.598649Z","shell.execute_reply.started":"2022-08-11T06:19:55.363907Z","shell.execute_reply":"2022-08-11T06:19:55.597690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"threshold = -0.2\n\nprediction_df = pd.DataFrame(data={'text':test_df['text'],'label':np.where(preds<threshold,0,1).ravel()})\n\nprediction_df.head(25)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:26:16.492322Z","iopub.execute_input":"2022-08-11T06:26:16.492707Z","iopub.status.idle":"2022-08-11T06:26:16.507181Z","shell.execute_reply.started":"2022-08-11T06:26:16.492666Z","shell.execute_reply":"2022-08-11T06:26:16.506094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission_df['target']=np.where(preds<threshold,0,1).ravel()\nsample_submission_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:26:31.053011Z","iopub.execute_input":"2022-08-11T06:26:31.053533Z","iopub.status.idle":"2022-08-11T06:26:31.066332Z","shell.execute_reply.started":"2022-08-11T06:26:31.053467Z","shell.execute_reply":"2022-08-11T06:26:31.065179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission_df.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:26:32.391752Z","iopub.execute_input":"2022-08-11T06:26:32.392041Z","iopub.status.idle":"2022-08-11T06:26:32.403722Z","shell.execute_reply.started":"2022-08-11T06:26:32.392011Z","shell.execute_reply":"2022-08-11T06:26:32.402689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}