{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <h><strong><center>⭐️⭐️Feedback Prize - Predicting Effective Arguments⭐️⭐️</center></strong></h>\n\n## 📢***First step of this competition : https://www.kaggle.com/code/venkatkumar001/nlpstarter-3-simple-ml-baseline-algo-with-kfold***\n\n\n<div>\n    <img class=\"marginauto\" src='https://t4.ftcdn.net/jpg/02/85/04/43/360_F_285044385_9rxqXa13GwQi8lgd2QB4hACgoGdp6LP5.jpg' alt=\"centered image\" />\n</div>\n\n## <strong><center>📢 Now! I'm trying transformer approach</center></strong>\n\n### ***📢 So that I'm trying DistilBert-Base-Cased(Hugging_face) using tensorflow ------------> Version 4***","metadata":{}},{"cell_type":"markdown","source":"# 📢 **Steps:**\n\n## **1. Import Necessary Library**\n\n## **2. Preprocessing**\n\n## **3. Initialize the Transformer - DistilBert-Base-cased ---------------> Refer version 4**\n\n## **3. Bert**\n\n## **4. Build the Model- Using Tensorflow_Framework**\n\n## **5. Predict Output**\n\n## **6. Generate Submission file**","metadata":{}},{"cell_type":"markdown","source":"# 📢 **Import Necessary Library**","metadata":{}},{"cell_type":"code","source":"import nltk\nfrom nltk.corpus import stopwords\nfrom nltk.stem import SnowballStemmer\nfrom string import punctuation\nfrom nltk.stem.wordnet import WordNetLemmatizer\nfrom tqdm import tqdm\nimport re\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport time\nimport datetime\nfrom scipy import sparse\nimport numpy as np\nimport pandas as pd\n\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-07-15T06:41:52.760347Z","iopub.execute_input":"2022-07-15T06:41:52.761147Z","iopub.status.idle":"2022-07-15T06:41:54.816821Z","shell.execute_reply.started":"2022-07-15T06:41:52.761009Z","shell.execute_reply":"2022-07-15T06:41:54.814507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" import datasets, transformers\n\nfrom transformers import TrainingArguments, Trainer\nfrom transformers import AutoModelForSequenceClassification, AutoTokenizer\nfrom transformers import AutoModelForMaskedLM\nos.environ[\"WANDB_DISABLED\"] = \"true\"\n\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Dense, Input, Dropout\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model\nfrom transformers import TFBertModel\nimport transformers","metadata":{"execution":{"iopub.status.busy":"2022-07-15T06:41:54.822774Z","iopub.execute_input":"2022-07-15T06:41:54.825570Z","iopub.status.idle":"2022-07-15T06:42:03.889616Z","shell.execute_reply.started":"2022-07-15T06:41:54.825530Z","shell.execute_reply":"2022-07-15T06:42:03.888679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📢 **Initialize the Configure**","metadata":{}},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\n# Configuration\nEPOCHS = 10\nBATCH_SIZE = 16\nMAX_LEN = 128","metadata":{"execution":{"iopub.status.busy":"2022-07-15T06:42:03.891056Z","iopub.execute_input":"2022-07-15T06:42:03.891814Z","iopub.status.idle":"2022-07-15T06:42:03.896621Z","shell.execute_reply.started":"2022-07-15T06:42:03.891777Z","shell.execute_reply":"2022-07-15T06:42:03.895594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📢 **Load Data and Display**","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/feedback-prize-effectiveness/train.csv')\ntest = pd.read_csv('../input/feedback-prize-effectiveness/test.csv')\nsample = pd.read_csv('../input/feedback-prize-effectiveness/sample_submission.csv')\nprint(f'Train_Shape: {train.shape},Test_Shape: {test.shape},Sample_Shape: {sample.shape}')\ndisplay(train.sample(2))\ndisplay(test.sample(2))\ndisplay(sample.sample(2))","metadata":{"execution":{"iopub.status.busy":"2022-07-15T06:42:03.899048Z","iopub.execute_input":"2022-07-15T06:42:03.899585Z","iopub.status.idle":"2022-07-15T06:42:04.220047Z","shell.execute_reply.started":"2022-07-15T06:42:03.899551Z","shell.execute_reply":"2022-07-15T06:42:04.218458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📢 **Simple Preprocessing - Just clean the text data outline**","metadata":{}},{"cell_type":"code","source":"def cleanup_text(text):\n    words = re.sub(pattern = '[^a-zA-Z]',repl = ' ', string = text)\n    words = words.lower()\n    return words\n\ncleanup_text('Every mountain speaks different ways!')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T06:42:04.221907Z","iopub.execute_input":"2022-07-15T06:42:04.222652Z","iopub.status.idle":"2022-07-15T06:42:04.234016Z","shell.execute_reply.started":"2022-07-15T06:42:04.222606Z","shell.execute_reply":"2022-07-15T06:42:04.232562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 📢 **Training Data**","metadata":{}},{"cell_type":"code","source":"text_preprocessed = train['discourse_text'].apply(cleanup_text)\ntext_preprocessed\ntrain['text_preprocessed'] = text_preprocessed\ndisplay(train.head())","metadata":{"execution":{"iopub.status.busy":"2022-07-15T06:42:04.235880Z","iopub.execute_input":"2022-07-15T06:42:04.236308Z","iopub.status.idle":"2022-07-15T06:42:05.102332Z","shell.execute_reply.started":"2022-07-15T06:42:04.236264Z","shell.execute_reply":"2022-07-15T06:42:05.101447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **About Transformer approach**\n\n<img src='https://miro.medium.com/max/1400/1*bSUO_Qib4te1xQmBlQjWaw.png'>","metadata":{}},{"cell_type":"markdown","source":"# 📢 **TransformerEncoder - Initialize the DistilBert-Base-Cased transformer---> Go and refer version 4 for this notebook** \n\n## 📢 **About distilbert-base**\n\nThe DistilBERT model was proposed in the blog post Smaller, faster, cheaper, lighter: Introducing DistilBERT, a distilled version of BERT, and the paper DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter. DistilBERT is a small, fast, cheap and light Transformer model trained by distilling BERT base. It has 40% less parameters than bert-base-uncased, runs 60% faster while preserving over 95% of BERT’s performances as measured on the GLUE language understanding benchmark.\n\n<img src='https://miro.medium.com/max/1400/1*uApTMl_f7eGdq_FDFc75SQ.png'>","metadata":{}},{"cell_type":"markdown","source":"# **Now i am try to build Bert base model**","metadata":{}},{"cell_type":"code","source":"# Texts, tokenizer inputs and Maxlength of inputs\ndef bert_encode(texts, tokenizer, max_len=MAX_LEN):\n    input_ids = []\n    token_type_ids = []\n    attention_mask = []\n    \n    for text in texts:\n        token = tokenizer(text, max_length=max_len, truncation=True, padding='max_length',\n                         add_special_tokens=True)\n        input_ids.append(token['input_ids'])\n        token_type_ids.append(token['token_type_ids'])\n        attention_mask.append(token['attention_mask'])\n    \n    return np.array(input_ids), np.array(token_type_ids), np.array(attention_mask)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T06:42:05.103996Z","iopub.execute_input":"2022-07-15T06:42:05.104464Z","iopub.status.idle":"2022-07-15T06:42:05.111569Z","shell.execute_reply.started":"2022-07-15T06:42:05.104411Z","shell.execute_reply":"2022-07-15T06:42:05.110361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# First load the real tokenizer\ntokenizer = transformers.BertTokenizer.from_pretrained('../input/huggingface-bert/bert-base-cased')\n# Save the loaded tokenizer locally\ntokenizer.save_pretrained('.')","metadata":{"execution":{"iopub.status.busy":"2022-07-15T06:42:05.113432Z","iopub.execute_input":"2022-07-15T06:42:05.114093Z","iopub.status.idle":"2022-07-15T06:42:05.386392Z","shell.execute_reply.started":"2022-07-15T06:42:05.114058Z","shell.execute_reply":"2022-07-15T06:42:05.385380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 📢 **Initialize the separation**","metadata":{}},{"cell_type":"code","source":"sep = tokenizer.sep_token\nsep","metadata":{"execution":{"iopub.status.busy":"2022-07-15T06:42:05.388155Z","iopub.execute_input":"2022-07-15T06:42:05.388552Z","iopub.status.idle":"2022-07-15T06:42:05.394581Z","shell.execute_reply.started":"2022-07-15T06:42:05.388506Z","shell.execute_reply":"2022-07-15T06:42:05.393664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 📢 **Generate New attribute: Combine discourse_type and clean text processed new attribute!Now just try without clean data version-14**","metadata":{}},{"cell_type":"code","source":"train['inputs'] = train.discourse_type + sep +train.discourse_text\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T06:42:05.400097Z","iopub.execute_input":"2022-07-15T06:42:05.400363Z","iopub.status.idle":"2022-07-15T06:42:05.431271Z","shell.execute_reply.started":"2022-07-15T06:42:05.400340Z","shell.execute_reply":"2022-07-15T06:42:05.430428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 📢 **Labeling the Target discourse_effectiveness data**","metadata":{}},{"cell_type":"code","source":"bin_map = {\"discourse_effectiveness\": {\"Ineffective\": 0, \"Adequate\": 1, \"Effective\": 2}}\ntrain = train.replace(bin_map)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T06:42:05.432684Z","iopub.execute_input":"2022-07-15T06:42:05.433073Z","iopub.status.idle":"2022-07-15T06:42:05.477469Z","shell.execute_reply.started":"2022-07-15T06:42:05.433046Z","shell.execute_reply":"2022-07-15T06:42:05.476558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T06:42:05.479011Z","iopub.execute_input":"2022-07-15T06:42:05.479352Z","iopub.status.idle":"2022-07-15T06:42:05.492259Z","shell.execute_reply.started":"2022-07-15T06:42:05.479319Z","shell.execute_reply":"2022-07-15T06:42:05.491274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📢 **Feature_Selection and Spliting the data**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_valid, y_train, y_valid = train_test_split(train['inputs'], train['discourse_effectiveness'], test_size=0.1, random_state=42)\n#X_train.shape,X_valid.shape,y_train.shape,y_valid.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-15T06:42:05.493628Z","iopub.execute_input":"2022-07-15T06:42:05.494248Z","iopub.status.idle":"2022-07-15T06:42:05.506989Z","shell.execute_reply.started":"2022-07-15T06:42:05.494212Z","shell.execute_reply":"2022-07-15T06:42:05.505924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 📢 **Convert string datatype**","metadata":{}},{"cell_type":"code","source":"X_train = bert_encode(X_train.astype(str), tokenizer)\nX_valid = bert_encode(X_valid.astype(str), tokenizer)\n\ny_train = y_train.values\ny_valid = y_valid.values","metadata":{"execution":{"iopub.status.busy":"2022-07-15T06:42:05.508481Z","iopub.execute_input":"2022-07-15T06:42:05.509017Z","iopub.status.idle":"2022-07-15T06:43:13.450267Z","shell.execute_reply.started":"2022-07-15T06:42:05.508981Z","shell.execute_reply":"2022-07-15T06:43:13.449229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📢 **Build the Model**","metadata":{}},{"cell_type":"markdown","source":"## ***Train and valid the data***","metadata":{}},{"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((X_train, y_train))\n    .repeat()\n    .shuffle(2048)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((X_valid, y_valid))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T06:43:13.451804Z","iopub.execute_input":"2022-07-15T06:43:13.452351Z","iopub.status.idle":"2022-07-15T06:43:19.093723Z","shell.execute_reply.started":"2022-07-15T06:43:13.452314Z","shell.execute_reply":"2022-07-15T06:43:19.092554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model(bert_model, max_len=MAX_LEN):    \n    input_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_ids\")\n    token_type_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"token_type_ids\")\n    attention_mask = Input(shape=(max_len,), dtype=tf.int32, name=\"attention_mask\")\n\n    sequence_output = bert_model(input_ids, token_type_ids=token_type_ids, attention_mask=attention_mask)[0]\n    clf_output = sequence_output[:, 0, :]\n    clf_output = Dropout(.1)(clf_output)\n    out = Dense(3, activation='softmax')(clf_output)\n    \n    model = Model(inputs=[input_ids, token_type_ids, attention_mask], outputs=out)\n    model.compile(Adam(lr=1e-4), loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-07-15T06:43:19.095271Z","iopub.execute_input":"2022-07-15T06:43:19.096202Z","iopub.status.idle":"2022-07-15T06:43:19.104820Z","shell.execute_reply.started":"2022-07-15T06:43:19.096164Z","shell.execute_reply":"2022-07-15T06:43:19.103554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntransformer_layer = (TFBertModel.from_pretrained('../input/huggingface-bert/bert-base-cased'))\nmodel = build_model(transformer_layer, max_len=MAX_LEN)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T06:43:19.106332Z","iopub.execute_input":"2022-07-15T06:43:19.106750Z","iopub.status.idle":"2022-07-15T06:43:32.973408Z","shell.execute_reply.started":"2022-07-15T06:43:19.106716Z","shell.execute_reply":"2022-07-15T06:43:32.972481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Model Plot**","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.utils import plot_model\n\nplot_model(model)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T06:43:32.974958Z","iopub.execute_input":"2022-07-15T06:43:32.976888Z","iopub.status.idle":"2022-07-15T06:43:34.061544Z","shell.execute_reply.started":"2022-07-15T06:43:32.976850Z","shell.execute_reply":"2022-07-15T06:43:34.060405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_history = model.fit(\n    train_dataset,\n    steps_per_epoch=200,\n    validation_data=valid_dataset,\n    epochs=EPOCHS\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T06:43:34.063206Z","iopub.execute_input":"2022-07-15T06:43:34.064258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📢 **Test Data - Prediction**","metadata":{}},{"cell_type":"code","source":"# Clean the text data\ntest_preprocessed = test['discourse_text'].apply(cleanup_text)\ntest_preprocessed\ntest['text_preprocessed'] =  test_preprocessed\ntest.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['processed'] = test.discourse_type + sep +test.discourse_text","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_processed = bert_encode(test.processed.astype(str), tokenizer)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📢 **Predict output**","metadata":{}},{"cell_type":"code","source":"preds = model.predict(test_processed, verbose=1)\npreds","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample['Ineffective'] = preds[:,0]\nsample['Adequate'] = preds[:,1]\nsample['Effective'] = preds[:,2]\nsample.sample(2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📢 **Generate CSV File**","metadata":{}},{"cell_type":"code","source":"sample.to_csv(\"submission.csv\", index=False)\nprint('Successfully executed!')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ⭐️⭐️**Thanks for visiting guys!**⭐️⭐️\n\n### ***Credits:***\n\n1. https://www.kaggle.com/code/imvision12/tensorflow-feedback-bert-baseline\n2. https://www.kaggle.com/code/venkatkumar001/nlpstarter-3-simple-ml-baseline-algo-with-kfold\n\n### ***If you know more about Huggingface transformer:***\n\n1. https://www.kaggle.com/code/venkatkumar001/nlp-starter2-hf-pretrain-finetune\n2. https://www.kaggle.com/code/venkatkumar001/transformeranatomy-encoder","metadata":{}}]}