{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-21T14:45:26.795542Z","iopub.execute_input":"2022-07-21T14:45:26.796088Z","iopub.status.idle":"2022-07-21T14:45:26.924145Z","shell.execute_reply.started":"2022-07-21T14:45:26.795960Z","shell.execute_reply":"2022-07-21T14:45:26.923082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install  \"tensorflow-text==2.8.*\"\n!pip install  tf-models-official==2.7.0","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:28:57.953112Z","iopub.execute_input":"2022-07-21T13:28:57.953562Z","iopub.status.idle":"2022-07-21T13:29:17.845708Z","shell.execute_reply.started":"2022-07-21T13:28:57.953527Z","shell.execute_reply":"2022-07-21T13:29:17.843960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_hub as hub\nimport tensorflow_text as text\nfrom official.nlp import optimization\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport spacy\nimport nltk\nimport re\nimport numpy as np\nfrom sklearn.metrics import classification_report\nimport seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:33:15.035081Z","iopub.execute_input":"2022-07-21T13:33:15.035524Z","iopub.status.idle":"2022-07-21T13:33:15.596585Z","shell.execute_reply.started":"2022-07-21T13:33:15.035494Z","shell.execute_reply":"2022-07-21T13:33:15.595065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv('/kaggle/input/nlp-getting-started/test.csv')\ntrain_data = pd.read_csv('/kaggle/input/nlp-getting-started/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:45:44.678840Z","iopub.execute_input":"2022-07-21T14:45:44.679210Z","iopub.status.idle":"2022-07-21T14:45:44.722167Z","shell.execute_reply.started":"2022-07-21T14:45:44.679179Z","shell.execute_reply":"2022-07-21T14:45:44.721269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:45:46.869693Z","iopub.execute_input":"2022-07-21T14:45:46.870031Z","iopub.status.idle":"2022-07-21T14:45:46.897225Z","shell.execute_reply.started":"2022-07-21T14:45:46.870002Z","shell.execute_reply":"2022-07-21T14:45:46.896310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Checking For Missing Values for every column**","metadata":{}},{"cell_type":"code","source":"for col in train_data.columns:\n  print(f'{col} : {len(train_data[train_data[col].isna()])/len(train_data)*100} %')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:29:58.793960Z","iopub.execute_input":"2022-07-21T13:29:58.794433Z","iopub.status.idle":"2022-07-21T13:29:58.810340Z","shell.execute_reply.started":"2022-07-21T13:29:58.794396Z","shell.execute_reply":"2022-07-21T13:29:58.809354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Fill missing values**","metadata":{}},{"cell_type":"code","source":"train_data.keyword.fillna(\"missing\", inplace=True)\ntrain_data.location.fillna(\"missing\", inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:32:11.198925Z","iopub.execute_input":"2022-07-21T13:32:11.199514Z","iopub.status.idle":"2022-07-21T13:32:11.207444Z","shell.execute_reply.started":"2022-07-21T13:32:11.199476Z","shell.execute_reply":"2022-07-21T13:32:11.206491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count_data = train_data['location'].value_counts().reset_index()[0:10]\ncount_data.rename(columns = {'index' : 'location', 'location' : 'count'}, inplace=True)\ncount_data","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:32:48.132818Z","iopub.execute_input":"2022-07-21T13:32:48.133207Z","iopub.status.idle":"2022-07-21T13:32:48.153340Z","shell.execute_reply.started":"2022-07-21T13:32:48.133177Z","shell.execute_reply":"2022-07-21T13:32:48.152531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,6))\nsns.barplot(x=count_data['location'], y= count_data['count'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:33:31.694663Z","iopub.execute_input":"2022-07-21T13:33:31.695121Z","iopub.status.idle":"2022-07-21T13:33:31.971699Z","shell.execute_reply.started":"2022-07-21T13:33:31.695085Z","shell.execute_reply":"2022-07-21T13:33:31.970637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_1 = train_data[train_data['target']==1]\ntext_0 = train_data[train_data['target']==0]\n\npos = text_1.groupby(['keyword']).count()['text'].sort_values(ascending=False)[:20]\nneg = text_0.groupby(['keyword']).count()['text'].sort_values(ascending=False)[:20]\n\npos = pos.reset_index()\npos.rename(columns={'text': 'count'}, inplace=True)\npos.head()\n\nneg = neg.reset_index()\nneg.rename(columns={'text': 'count'}, inplace=True)\nneg.head()\n\n\nfig,ax = plt.subplots(1,2, figsize=(25,10))\nsns.barplot(y=pos['keyword'], x=pos['count'],ax=ax[0])\nsns.barplot(y=neg['keyword'], x=neg['count'],ax=ax[1])\n\n\nfor ax in [ax[0], ax[1]]:\n    \n    ax.set_title('Number of tweets per keyword', fontsize = 20)\n    \n    ax.set_ylabel('') \n    ax.set_xlabel('')\n\n    ax.set_yticklabels(labels =ax.get_yticklabels(),fontsize = 15)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:34:59.319909Z","iopub.execute_input":"2022-07-21T13:34:59.320435Z","iopub.status.idle":"2022-07-21T13:34:59.971738Z","shell.execute_reply.started":"2022-07-21T13:34:59.320399Z","shell.execute_reply":"2022-07-21T13:34:59.970510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"!python -m spacy download en_core_web_sm","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:35:44.624958Z","iopub.execute_input":"2022-07-21T13:35:44.625419Z","iopub.status.idle":"2022-07-21T13:35:57.058155Z","shell.execute_reply.started":"2022-07-21T13:35:44.625383Z","shell.execute_reply":"2022-07-21T13:35:57.056771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Clean Text Data**","metadata":{}},{"cell_type":"code","source":"nlp = spacy.load(\"en_core_web_sm\")\nfrom nltk.corpus import stopwords\n\nnltk.download('stopwords')\nnltk.download('punkt')\nnltk_stop = stopwords.words('english')\nnltk_stop.remove('not')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:35:58.531374Z","iopub.execute_input":"2022-07-21T13:35:58.531818Z","iopub.status.idle":"2022-07-21T13:35:59.190511Z","shell.execute_reply.started":"2022-07-21T13:35:58.531780Z","shell.execute_reply":"2022-07-21T13:35:59.189433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_text(text):\n  text = text.lower()\n  text = re.sub('[a-z]*[:.]+\\S+', '', text)\n  text = re.sub('[^a-zA-Z0-9]', '', text)\n  return text\n\ndef process_tweets(row_text):\n  words = []\n  for token in nlp(row_text):\n    if token.text not in nltk_stop:\n      words.append(clean_text(token.lemma_))\n  return \" \".join(words)\n ","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:36:21.863351Z","iopub.execute_input":"2022-07-21T13:36:21.864911Z","iopub.status.idle":"2022-07-21T13:36:21.873450Z","shell.execute_reply.started":"2022-07-21T13:36:21.864845Z","shell.execute_reply":"2022-07-21T13:36:21.871961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['text'] =  test_data['text'].apply(lambda x : process_tweets(x))\ntrain_data['text'] =  train_data['text'].apply(lambda x : process_tweets(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:36:36.929624Z","iopub.execute_input":"2022-07-21T13:36:36.930801Z","iopub.status.idle":"2022-07-21T13:39:00.421200Z","shell.execute_reply.started":"2022-07-21T13:36:36.930750Z","shell.execute_reply":"2022-07-21T13:39:00.419213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds = train_data[['text', 'target']]\ntrain_ds","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:39:00.445592Z","iopub.execute_input":"2022-07-21T13:39:00.446448Z","iopub.status.idle":"2022-07-21T13:39:00.464319Z","shell.execute_reply.started":"2022-07-21T13:39:00.446324Z","shell.execute_reply":"2022-07-21T13:39:00.463067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Load Bert Model **","metadata":{}},{"cell_type":"code","source":"bert_model_name = 'small_bert/bert_en_uncased_L-4_H-512_A-8'\ntfhub_handle_encoder = 'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-4_H-512_A-8/1'\ntfhub_handle_preprocess = 'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3'","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:39:00.467222Z","iopub.execute_input":"2022-07-21T13:39:00.467612Z","iopub.status.idle":"2022-07-21T13:39:00.475669Z","shell.execute_reply.started":"2022-07-21T13:39:00.467577Z","shell.execute_reply":"2022-07-21T13:39:00.474800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bert_model = hub.KerasLayer(tfhub_handle_encoder)\nbert_preprocess_model = hub.KerasLayer(tfhub_handle_preprocess)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:39:00.476879Z","iopub.execute_input":"2022-07-21T13:39:00.477377Z","iopub.status.idle":"2022-07-21T13:39:11.484419Z","shell.execute_reply.started":"2022-07-21T13:39:00.477335Z","shell.execute_reply":"2022-07-21T13:39:11.483035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_text = list(train_data[0:1].text)[0]\nsample_text","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:40:51.171285Z","iopub.execute_input":"2022-07-21T13:40:51.171749Z","iopub.status.idle":"2022-07-21T13:40:51.179813Z","shell.execute_reply.started":"2022-07-21T13:40:51.171711Z","shell.execute_reply":"2022-07-21T13:40:51.178593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"processed_text = bert_preprocess_model([sample_text])","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:40:53.196489Z","iopub.execute_input":"2022-07-21T13:40:53.196978Z","iopub.status.idle":"2022-07-21T13:40:53.204259Z","shell.execute_reply.started":"2022-07-21T13:40:53.196936Z","shell.execute_reply":"2022-07-21T13:40:53.202947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'keys :  {list(processed_text.keys())}')\nprint(f'input_word_id :  {processed_text[\"input_word_ids\"]}')\nprint(f'type_id :  {processed_text[\"input_type_ids\"]}')\nprint(f'input_mask :  {processed_text[\"input_mask\"]}')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:40:56.025003Z","iopub.execute_input":"2022-07-21T13:40:56.025438Z","iopub.status.idle":"2022-07-21T13:40:56.036500Z","shell.execute_reply.started":"2022-07-21T13:40:56.025404Z","shell.execute_reply":"2022-07-21T13:40:56.035272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bert_results = bert_model(processed_text)\n#bert_results[\"pooled_output\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:41:20.467166Z","iopub.execute_input":"2022-07-21T13:41:20.468053Z","iopub.status.idle":"2022-07-21T13:41:20.538067Z","shell.execute_reply.started":"2022-07-21T13:41:20.468009Z","shell.execute_reply":"2022-07-21T13:41:20.537039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list(bert_results.keys())","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:41:32.366865Z","iopub.execute_input":"2022-07-21T13:41:32.367275Z","iopub.status.idle":"2022-07-21T13:41:32.374415Z","shell.execute_reply.started":"2022-07-21T13:41:32.367241Z","shell.execute_reply":"2022-07-21T13:41:32.373430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_classifier_model():\n  text_input = tf.keras.layers.Input(shape=(), dtype=tf.string, name='text')\n  processing_layer = hub.KerasLayer(tfhub_handle_preprocess, name='preprocess_text')\n  encoder_input = processing_layer(text_input)\n  encoder = hub.KerasLayer(tfhub_handle_encoder, trainable=True, name='BertEncoders')\n  outputs = encoder(encoder_input)\n  net = outputs['pooled_output']\n  # sequence_output =  outputs['sequence_output']\n  # net = sequence_output[:, 0, :]\n  net = tf.keras.layers.Dense(32, activation= 'relu')(net)\n  net = tf.keras.layers.Dropout(.1)(net)\n  net = tf.keras.layers.Dense(1, activation='sigmoid', name='classifier')(net)\n  return tf.keras.Model(text_input, net)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:41:45.421904Z","iopub.execute_input":"2022-07-21T13:41:45.422311Z","iopub.status.idle":"2022-07-21T13:41:45.430589Z","shell.execute_reply.started":"2022-07-21T13:41:45.422279Z","shell.execute_reply":"2022-07-21T13:41:45.429538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classifier_model = build_classifier_model()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:41:55.813986Z","iopub.execute_input":"2022-07-21T13:41:55.814412Z","iopub.status.idle":"2022-07-21T13:42:05.242403Z","shell.execute_reply.started":"2022-07-21T13:41:55.814375Z","shell.execute_reply":"2022-07-21T13:42:05.241175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(classifier_model)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:42:06.449965Z","iopub.execute_input":"2022-07-21T13:42:06.450454Z","iopub.status.idle":"2022-07-21T13:42:07.690105Z","shell.execute_reply.started":"2022-07-21T13:42:06.450413Z","shell.execute_reply":"2022-07-21T13:42:07.688157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss = tf.keras.losses.BinaryCrossentropy(from_logits=True)\nmetrics = tf.metrics.BinaryAccuracy()\ncallbacks = tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:43:32.673878Z","iopub.execute_input":"2022-07-21T13:43:32.674348Z","iopub.status.idle":"2022-07-21T13:43:32.686110Z","shell.execute_reply.started":"2022-07-21T13:43:32.674299Z","shell.execute_reply":"2022-07-21T13:43:32.684946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classifier_model.compile(optimizer = tf.keras.optimizers.Adam(lr = 0.0001, \n                                                              beta_1=0.9, beta_2=0.999, epsilon=1e-07), loss=loss, metrics=metrics)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:43:35.233969Z","iopub.execute_input":"2022-07-21T13:43:35.234415Z","iopub.status.idle":"2022-07-21T13:43:35.251358Z","shell.execute_reply.started":"2022-07-21T13:43:35.234380Z","shell.execute_reply":"2022-07-21T13:43:35.250286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(train_ds.text,train_ds.target, test_size=.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:43:38.717644Z","iopub.execute_input":"2022-07-21T13:43:38.718112Z","iopub.status.idle":"2022-07-21T13:43:38.730969Z","shell.execute_reply.started":"2022-07-21T13:43:38.718076Z","shell.execute_reply":"2022-07-21T13:43:38.729924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = classifier_model.fit(x=X_train,y=y_train,  validation_data = (X_test, y_test), epochs = 32,batch_size=64, callbacks=callbacks)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T13:43:50.402830Z","iopub.execute_input":"2022-07-21T13:43:50.403295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classifier_model.evaluate(X_test,y_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_p = classifier_model.predict(X_test)\ny_p = np.where(y_p > .5 , 1, 0)\nprint(classification_report(y_test, y_p))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data[['text']].head()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:45:57.541425Z","iopub.execute_input":"2022-07-21T14:45:57.541753Z","iopub.status.idle":"2022-07-21T14:45:57.555264Z","shell.execute_reply.started":"2022-07-21T14:45:57.541726Z","shell.execute_reply":"2022-07-21T14:45:57.554227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:46:28.403282Z","iopub.execute_input":"2022-07-21T14:46:28.403649Z","iopub.status.idle":"2022-07-21T14:46:28.420551Z","shell.execute_reply.started":"2022-07-21T14:46:28.403618Z","shell.execute_reply":"2022-07-21T14:46:28.419361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data[['text']].head()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:45:57.541425Z","iopub.execute_input":"2022-07-21T14:45:57.541753Z","iopub.status.idle":"2022-07-21T14:45:57.555264Z","shell.execute_reply.started":"2022-07-21T14:45:57.541726Z","shell.execute_reply":"2022-07-21T14:45:57.554227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ythat = classifier_model.predict(test_data.text)\ntest_data['pred'] = ythat","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def map_class(x):\n  if x >= .5:\n    return 1\n  else:\n     return 0","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['target'] = test_data['pred'].apply(lambda x : map_class(x))\ntest_data[['id', 'target']].to_csv('bert_res.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}