{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# loading all dependencies","metadata":{}},{"cell_type":"code","source":"import re\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nimport pandas as pd\n\nimport tensorflow as tf \n# from tensorflow.keras.preprocessing.text import Tokenizer\n# from tensorflow.keras.preprocessing.sequence import pad_sequences\n# from tensorflow.keras.layers import LSTM, Embedding, Dense, Bidirectional, Flatten, GlobalAveragePooling1D\n# from tensorflow.keras import Sequential\n# from keras.callbacks import EarlyStopping\nfrom sklearn.model_selection import train_test_split\n\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize\nstop_words = set(stopwords.words('english'))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-05T10:02:44.645091Z","iopub.execute_input":"2022-07-05T10:02:44.645449Z","iopub.status.idle":"2022-07-05T10:02:44.934872Z","shell.execute_reply.started":"2022-07-05T10:02:44.645419Z","shell.execute_reply":"2022-07-05T10:02:44.933888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# A dependency of the preprocessing for BERT inputs","metadata":{}},{"cell_type":"markdown","source":"# About BERT\n**BERT and other Transformer encoder architectures have been wildly successful on a variety**\n\n**of tasks in NLP (natural language processing). They compute vector-space representations of natural language**\n\n**that are suitable for use in deep learning models.**","metadata":{}},{"cell_type":"code","source":"!pip install \"tensorflow-text==2.8.*\"","metadata":{"execution":{"iopub.status.busy":"2022-07-05T10:00:29.984479Z","iopub.execute_input":"2022-07-05T10:00:29.985167Z","iopub.status.idle":"2022-07-05T10:02:21.154896Z","shell.execute_reply.started":"2022-07-05T10:00:29.985130Z","shell.execute_reply":"2022-07-05T10:02:21.153721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow_text as text","metadata":{"execution":{"iopub.status.busy":"2022-07-05T10:02:35.779038Z","iopub.execute_input":"2022-07-05T10:02:35.780429Z","iopub.status.idle":"2022-07-05T10:02:35.786183Z","shell.execute_reply.started":"2022-07-05T10:02:35.780390Z","shell.execute_reply":"2022-07-05T10:02:35.785236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow_hub as hub","metadata":{"execution":{"iopub.status.busy":"2022-07-05T10:04:28.306182Z","iopub.execute_input":"2022-07-05T10:04:28.307102Z","iopub.status.idle":"2022-07-05T10:04:28.312466Z","shell.execute_reply.started":"2022-07-05T10:04:28.307046Z","shell.execute_reply":"2022-07-05T10:04:28.311411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Laoding data","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/feedback-prize-effectiveness/train.csv')\ndt = pd.read_csv('../input/feedback-prize-effectiveness/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T10:02:48.144425Z","iopub.execute_input":"2022-07-05T10:02:48.144858Z","iopub.status.idle":"2022-07-05T10:02:48.446875Z","shell.execute_reply.started":"2022-07-05T10:02:48.144827Z","shell.execute_reply":"2022-07-05T10:02:48.445875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# visualizing date","metadata":{}},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T10:02:51.114318Z","iopub.execute_input":"2022-07-05T10:02:51.115037Z","iopub.status.idle":"2022-07-05T10:02:51.135574Z","shell.execute_reply.started":"2022-07-05T10:02:51.114980Z","shell.execute_reply":"2022-07-05T10:02:51.134696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dt.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T10:02:52.024053Z","iopub.execute_input":"2022-07-05T10:02:52.024720Z","iopub.status.idle":"2022-07-05T10:02:52.035607Z","shell.execute_reply.started":"2022-07-05T10:02:52.024664Z","shell.execute_reply":"2022-07-05T10:02:52.034530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T10:02:52.754354Z","iopub.execute_input":"2022-07-05T10:02:52.755237Z","iopub.status.idle":"2022-07-05T10:02:52.936337Z","shell.execute_reply.started":"2022-07-05T10:02:52.755197Z","shell.execute_reply":"2022-07-05T10:02:52.935257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,10))\nsns.countplot(df['discourse_type'])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T10:02:53.634138Z","iopub.execute_input":"2022-07-05T10:02:53.634723Z","iopub.status.idle":"2022-07-05T10:02:53.890566Z","shell.execute_reply.started":"2022-07-05T10:02:53.634667Z","shell.execute_reply":"2022-07-05T10:02:53.889603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# data preprocessing ","metadata":{}},{"cell_type":"code","source":"def creat_text(_id):\n    t = open(f'/kaggle/input/feedback-prize-effectiveness/train/{_id}.txt').read()\n    return t        \n        \ndf['text'] = df['essay_id'].apply(lambda x: creat_text(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-05T10:02:55.994345Z","iopub.execute_input":"2022-07-05T10:02:55.995230Z","iopub.status.idle":"2022-07-05T10:03:37.843383Z","shell.execute_reply.started":"2022-07-05T10:02:55.995192Z","shell.execute_reply":"2022-07-05T10:03:37.842412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head(20)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T10:03:37.845978Z","iopub.execute_input":"2022-07-05T10:03:37.847350Z","iopub.status.idle":"2022-07-05T10:03:37.863142Z","shell.execute_reply.started":"2022-07-05T10:03:37.847306Z","shell.execute_reply":"2022-07-05T10:03:37.862102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Looking for some Text and cleaning it..","metadata":{}},{"cell_type":"code","source":"df['text'][15]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T10:03:44.754939Z","iopub.execute_input":"2022-07-05T10:03:44.755564Z","iopub.status.idle":"2022-07-05T10:03:44.762822Z","shell.execute_reply.started":"2022-07-05T10:03:44.755523Z","shell.execute_reply":"2022-07-05T10:03:44.761618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cleanup_text(text):\n    words = re.sub(pattern = '[^a-zA-Z]',repl = ' ', string = text)\n    words = words.lower()\n    return words\n\ndf['CleanText'] = df['text'].apply(lambda x: cleanup_text(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-05T10:03:46.054975Z","iopub.execute_input":"2022-07-05T10:03:46.055576Z","iopub.status.idle":"2022-07-05T10:03:53.796135Z","shell.execute_reply.started":"2022-07-05T10:03:46.055542Z","shell.execute_reply":"2022-07-05T10:03:53.795120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T10:03:53.799764Z","iopub.execute_input":"2022-07-05T10:03:53.800055Z","iopub.status.idle":"2022-07-05T10:03:53.812405Z","shell.execute_reply.started":"2022-07-05T10:03:53.800023Z","shell.execute_reply":"2022-07-05T10:03:53.811283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating data ready for training","metadata":{}},{"cell_type":"code","source":"y = pd.get_dummies(df['discourse_effectiveness'])\ny","metadata":{"execution":{"iopub.status.busy":"2022-07-05T10:03:56.504133Z","iopub.execute_input":"2022-07-05T10:03:56.504961Z","iopub.status.idle":"2022-07-05T10:03:56.522327Z","shell.execute_reply.started":"2022-07-05T10:03:56.504924Z","shell.execute_reply":"2022-07-05T10:03:56.521360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df['CleanText']","metadata":{"execution":{"iopub.status.busy":"2022-07-05T10:03:58.274135Z","iopub.execute_input":"2022-07-05T10:03:58.276789Z","iopub.status.idle":"2022-07-05T10:03:58.281578Z","shell.execute_reply.started":"2022-07-05T10:03:58.276744Z","shell.execute_reply":"2022-07-05T10:03:58.280398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Train test split**","metadata":{}},{"cell_type":"code","source":"X_train,X_test,y_train,y_test = train_test_split(X, y, random_state=42, test_size=0.15, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T10:03:59.694854Z","iopub.execute_input":"2022-07-05T10:03:59.695476Z","iopub.status.idle":"2022-07-05T10:03:59.709720Z","shell.execute_reply.started":"2022-07-05T10:03:59.695437Z","shell.execute_reply":"2022-07-05T10:03:59.708794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape,X_test.shape,y_train.shape,y_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-05T10:04:01.876460Z","iopub.execute_input":"2022-07-05T10:04:01.876843Z","iopub.status.idle":"2022-07-05T10:04:01.884296Z","shell.execute_reply.started":"2022-07-05T10:04:01.876812Z","shell.execute_reply":"2022-07-05T10:04:01.883158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading models from TensorFlow Hub","metadata":{}},{"cell_type":"code","source":"tfhub_handle_preprocess = \"https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/1\"\ntfhub_handle_encoder = \"https://tfhub.dev/tensorflow/bert_en_uncased_L-12_H-768_A-12/3\"","metadata":{"execution":{"iopub.status.busy":"2022-07-05T10:04:14.674944Z","iopub.execute_input":"2022-07-05T10:04:14.675839Z","iopub.status.idle":"2022-07-05T10:04:14.680111Z","shell.execute_reply.started":"2022-07-05T10:04:14.675803Z","shell.execute_reply":"2022-07-05T10:04:14.678842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# The preprocessing model\n**Text inputs need to be transformed to numeric token ids and arranged in several \nTensors before being input to BERT. \nTensorFlow Hub provides a matching preprocessing model for each of the BERT**","metadata":{}},{"cell_type":"markdown","source":"# Using the BERT model","metadata":{}},{"cell_type":"code","source":"bert_preprocess_model = hub.KerasLayer(tfhub_handle_preprocess)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T10:04:36.994747Z","iopub.execute_input":"2022-07-05T10:04:36.995421Z","iopub.status.idle":"2022-07-05T10:04:43.693309Z","shell.execute_reply.started":"2022-07-05T10:04:36.995379Z","shell.execute_reply":"2022-07-05T10:04:43.692320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_test = ['this is such an amazing movie!']\ntext_preprocessed = bert_preprocess_model(text_test)\n\nprint(f'Keys       : {list(text_preprocessed.keys())}')\nprint(f'Shape      : {text_preprocessed[\"input_word_ids\"].shape}')\nprint(f'Word Ids   : {text_preprocessed[\"input_word_ids\"][0, :12]}')\nprint(f'Input Mask : {text_preprocessed[\"input_mask\"][0, :12]}')\nprint(f'Type Ids   : {text_preprocessed[\"input_type_ids\"][0, :12]}')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T10:04:46.604338Z","iopub.execute_input":"2022-07-05T10:04:46.605287Z","iopub.status.idle":"2022-07-05T10:04:47.077714Z","shell.execute_reply.started":"2022-07-05T10:04:46.605240Z","shell.execute_reply":"2022-07-05T10:04:47.076740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define your model","metadata":{}},{"cell_type":"code","source":"def build_classifier_model():\n    text_input = tf.keras.layers.Input(shape=(), dtype=tf.string, name='text')\n    preprocessing_layer = hub.KerasLayer(tfhub_handle_preprocess, name='preprocessing')\n    encoder_inputs = preprocessing_layer(text_input)\n    encoder = hub.KerasLayer(tfhub_handle_encoder, trainable=True, name='BERT_encoder')\n    outputs = encoder(encoder_inputs)\n    net = outputs['pooled_output']\n    net = tf.keras.layers.Dropout(0.1)(net)\n    #net = tf.keras.layers.Dense(3, activation=None, name='classifier')(net)\n    net = tf.keras.layers.Dense(3, activation='softmax', name='classifier')(net)\n    return tf.keras.Model(text_input, net)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T10:04:50.704348Z","iopub.execute_input":"2022-07-05T10:04:50.704769Z","iopub.status.idle":"2022-07-05T10:04:50.712243Z","shell.execute_reply.started":"2022-07-05T10:04:50.704733Z","shell.execute_reply":"2022-07-05T10:04:50.711149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Let's take a look at the model's structure.**","metadata":{}},{"cell_type":"markdown","source":"![image](https://www.tensorflow.org/static/text/tutorials/classify_text_with_bert_files/output_0EmzyHZXKIpm_0.png)","metadata":{}},{"cell_type":"code","source":"classifier_model = build_classifier_model()\nbert_raw_result = classifier_model(tf.constant(text_test))\nprint(tf.sigmoid(bert_raw_result))","metadata":{"execution":{"iopub.status.busy":"2022-07-05T10:04:54.144172Z","iopub.execute_input":"2022-07-05T10:04:54.145087Z","iopub.status.idle":"2022-07-05T10:05:32.484404Z","shell.execute_reply.started":"2022-07-05T10:04:54.145053Z","shell.execute_reply":"2022-07-05T10:05:32.482916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss = tf.keras.losses.BinaryCrossentropy(from_logits=True)\n# loss = tf.keras.losses.SparseCategoricalCrossentropy (from_logits=True)\nmetrics = tf.metrics.BinaryAccuracy()\nclassifier_model.compile(optimizer='adam',loss=loss,metrics=metrics)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T10:05:51.294159Z","iopub.execute_input":"2022-07-05T10:05:51.294864Z","iopub.status.idle":"2022-07-05T10:05:51.313944Z","shell.execute_reply.started":"2022-07-05T10:05:51.294827Z","shell.execute_reply":"2022-07-05T10:05:51.313045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = 2\nhistory = classifier_model.fit(X_train,y_train, validation_data=(X_test,y_test),epochs=epochs)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T10:05:52.514158Z","iopub.execute_input":"2022-07-05T10:05:52.514519Z","iopub.status.idle":"2022-07-05T10:27:29.367598Z","shell.execute_reply.started":"2022-07-05T10:05:52.514487Z","shell.execute_reply":"2022-07-05T10:27:29.366491Z"},"trusted":true},"execution_count":null,"outputs":[]}]}