{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import nltk\nfrom nltk.corpus import stopwords\nfrom nltk.stem import SnowballStemmer\nfrom string import punctuation\nfrom nltk.stem.wordnet import WordNetLemmatizer\nfrom tqdm import tqdm\nimport re\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport time\nimport datetime\nfrom scipy import sparse\nimport numpy as np\nimport pandas as pd\n\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:53:53.054118Z","iopub.execute_input":"2022-08-03T09:53:53.054569Z","iopub.status.idle":"2022-08-03T09:53:54.157225Z","shell.execute_reply.started":"2022-08-03T09:53:53.054476Z","shell.execute_reply":"2022-08-03T09:53:54.156258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" import datasets, transformers\n\nfrom transformers import TrainingArguments, Trainer\nfrom transformers import AutoModelForSequenceClassification, AutoTokenizer\nfrom transformers import AutoModelForMaskedLM\nos.environ[\"WANDB_DISABLED\"] = \"true\"\n\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Dense, Input, Dropout\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model\nfrom transformers import TFBertModel\nimport transformers","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:53:55.824710Z","iopub.execute_input":"2022-08-03T09:53:55.825723Z","iopub.status.idle":"2022-08-03T09:54:03.627280Z","shell.execute_reply.started":"2022-08-03T09:53:55.825680Z","shell.execute_reply":"2022-08-03T09:54:03.626316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📢 **Initialize the Configure**","metadata":{}},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\n# Configuration\n#EPOCHS = 10\nBATCH_SIZE =56\nMAX_LEN = 128","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:54:08.289449Z","iopub.execute_input":"2022-08-03T09:54:08.290211Z","iopub.status.idle":"2022-08-03T09:54:08.300152Z","shell.execute_reply.started":"2022-08-03T09:54:08.290179Z","shell.execute_reply":"2022-08-03T09:54:08.295585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📢 **Load Data and Display**","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/feedback-prize-effectiveness/train.csv')\ntest = pd.read_csv('../input/feedback-prize-effectiveness/test.csv')\nsample = pd.read_csv('../input/feedback-prize-effectiveness/sample_submission.csv')\nprint(f'Train_Shape: {train.shape},Test_Shape: {test.shape},Sample_Shape: {sample.shape}')\ndisplay(train.sample(2))\ndisplay(test.sample(2))\ndisplay(sample.sample(2))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:54:09.588867Z","iopub.execute_input":"2022-08-03T09:54:09.589293Z","iopub.status.idle":"2022-08-03T09:54:09.911330Z","shell.execute_reply.started":"2022-08-03T09:54:09.589253Z","shell.execute_reply":"2022-08-03T09:54:09.910308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📢 **Simple Preprocessing - Just clean the text data outline**","metadata":{}},{"cell_type":"code","source":"def cleanup_text(text):\n    words = re.sub(pattern = '[^a-zA-Z]',repl = ' ', string = text)\n    words = words.lower()\n    return words\n\ncleanup_text('Every mountain speaks different ways!')\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:54:15.148908Z","iopub.execute_input":"2022-08-03T09:54:15.149259Z","iopub.status.idle":"2022-08-03T09:54:15.156744Z","shell.execute_reply.started":"2022-08-03T09:54:15.149229Z","shell.execute_reply":"2022-08-03T09:54:15.155667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 📢 **Training Data**","metadata":{}},{"cell_type":"code","source":"text_preprocessed = train['discourse_text'].apply(cleanup_text)\ntext_preprocessed\ntrain['text_preprocessed'] = text_preprocessed\ndisplay(train.head())","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:54:17.960716Z","iopub.execute_input":"2022-08-03T09:54:17.961793Z","iopub.status.idle":"2022-08-03T09:54:18.502326Z","shell.execute_reply.started":"2022-08-03T09:54:17.961748Z","shell.execute_reply":"2022-08-03T09:54:18.501274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Now i am try to build Bert base model**","metadata":{}},{"cell_type":"code","source":"# Texts, tokenizer inputs and Maxlength of inputs\ndef bert_encode(texts, tokenizer, max_len=MAX_LEN):\n    input_ids = []\n    token_type_ids = []\n    attention_mask = []\n    \n    for text in texts:\n        token = tokenizer(text, max_length=max_len, truncation=True, padding='max_length',\n                         add_special_tokens=True)\n        input_ids.append(token['input_ids'])\n        token_type_ids.append(token['token_type_ids'])\n        attention_mask.append(token['attention_mask'])\n    \n    return np.array(input_ids), np.array(token_type_ids), np.array(attention_mask)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:56:40.576540Z","iopub.execute_input":"2022-08-02T07:56:40.576892Z","iopub.status.idle":"2022-08-02T07:56:40.583504Z","shell.execute_reply.started":"2022-08-02T07:56:40.576862Z","shell.execute_reply":"2022-08-02T07:56:40.582301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# First load the real tokenizer\ntokenizer = transformers.BertTokenizer.from_pretrained('../input/huggingface-bert/bert-base-cased')\n# Save the loaded tokenizer locally\ntokenizer.save_pretrained('.')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:56:46.794793Z","iopub.execute_input":"2022-08-02T07:56:46.795137Z","iopub.status.idle":"2022-08-02T07:56:47.058314Z","shell.execute_reply.started":"2022-08-02T07:56:46.795107Z","shell.execute_reply":"2022-08-02T07:56:47.057390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 📢 **Initialize the separation**","metadata":{}},{"cell_type":"code","source":"sep = tokenizer.sep_token\nsep","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:56:53.163873Z","iopub.execute_input":"2022-08-02T07:56:53.164309Z","iopub.status.idle":"2022-08-02T07:56:53.171282Z","shell.execute_reply.started":"2022-08-02T07:56:53.164264Z","shell.execute_reply":"2022-08-02T07:56:53.170261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 📢 **Generate New attribute: Combine discourse_type and clean text processed new attribute!Now just try without clean data version-14**","metadata":{}},{"cell_type":"code","source":"train['inputs'] = train.discourse_type + sep +train.discourse_text\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:56:58.979751Z","iopub.execute_input":"2022-08-02T07:56:58.980331Z","iopub.status.idle":"2022-08-02T07:56:59.029656Z","shell.execute_reply.started":"2022-08-02T07:56:58.980286Z","shell.execute_reply":"2022-08-02T07:56:59.028629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 📢 **Labeling the Target discourse_effectiveness data**","metadata":{}},{"cell_type":"code","source":"bin_map = {\"discourse_effectiveness\": {\"Ineffective\": 0, \"Adequate\": 1, \"Effective\": 2}}\ntrain = train.replace(bin_map)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:57:03.895357Z","iopub.execute_input":"2022-08-02T07:57:03.895710Z","iopub.status.idle":"2022-08-02T07:57:03.934324Z","shell.execute_reply.started":"2022-08-02T07:57:03.895682Z","shell.execute_reply":"2022-08-02T07:57:03.933332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:57:07.797798Z","iopub.execute_input":"2022-08-02T07:57:07.798159Z","iopub.status.idle":"2022-08-02T07:57:07.811880Z","shell.execute_reply.started":"2022-08-02T07:57:07.798128Z","shell.execute_reply":"2022-08-02T07:57:07.810947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📢 **Feature_Selection and Spliting the data**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_valid, y_train, y_valid = train_test_split(train['inputs'], train['discourse_effectiveness'], test_size=0.1, random_state=42)\n#X_train.shape,X_valid.shape,y_train.shape,y_valid.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:57:13.399517Z","iopub.execute_input":"2022-08-02T07:57:13.399882Z","iopub.status.idle":"2022-08-02T07:57:13.412855Z","shell.execute_reply.started":"2022-08-02T07:57:13.399853Z","shell.execute_reply":"2022-08-02T07:57:13.411958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 📢 **Convert string datatype**","metadata":{}},{"cell_type":"code","source":"X_train = bert_encode(X_train.astype(str), tokenizer)\nX_valid = bert_encode(X_valid.astype(str), tokenizer)\n\ny_train = y_train.values\ny_valid = y_valid.values","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:57:21.294391Z","iopub.execute_input":"2022-08-02T07:57:21.294848Z","iopub.status.idle":"2022-08-02T07:58:17.618030Z","shell.execute_reply.started":"2022-08-02T07:57:21.294808Z","shell.execute_reply":"2022-08-02T07:58:17.616990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:58:37.658352Z","iopub.execute_input":"2022-08-02T07:58:37.659324Z","iopub.status.idle":"2022-08-02T07:58:37.668237Z","shell.execute_reply.started":"2022-08-02T07:58:37.659279Z","shell.execute_reply":"2022-08-02T07:58:37.667078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📢 **Build the Model**","metadata":{}},{"cell_type":"markdown","source":"## ***Train and valid the data***","metadata":{}},{"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((X_train, y_train))\n    .repeat()\n    .shuffle(2048)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((X_valid, y_valid))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T05:41:01.545741Z","iopub.execute_input":"2022-08-01T05:41:01.546496Z","iopub.status.idle":"2022-08-01T05:41:06.885119Z","shell.execute_reply.started":"2022-08-01T05:41:01.546456Z","shell.execute_reply":"2022-08-01T05:41:06.884169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model(bert_model, max_len=MAX_LEN):    \n    input_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_ids\")\n    token_type_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"token_type_ids\")\n    attention_mask = Input(shape=(max_len,), dtype=tf.int32, name=\"attention_mask\")\n\n    sequence_output = bert_model(input_ids, token_type_ids=token_type_ids, attention_mask=attention_mask)[0]\n    clf_output = sequence_output[:, 0, :]\n    clf_output = Dropout(.1)(clf_output)\n    out = Dense(3, activation='softmax')(clf_output)\n    \n    model = Model(inputs=[input_ids, token_type_ids, attention_mask], outputs=out)\n    model.compile(Adam(lr=1e-4), loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-08-01T05:41:06.886417Z","iopub.execute_input":"2022-08-01T05:41:06.887021Z","iopub.status.idle":"2022-08-01T05:41:06.896130Z","shell.execute_reply.started":"2022-08-01T05:41:06.886985Z","shell.execute_reply":"2022-08-01T05:41:06.895103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntransformer_layer = (TFBertModel.from_pretrained('../input/huggingface-bert/bert-base-cased'))\nmodel = build_model(transformer_layer, max_len=MAX_LEN)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T05:41:06.897474Z","iopub.execute_input":"2022-08-01T05:41:06.897999Z","iopub.status.idle":"2022-08-01T05:41:22.046292Z","shell.execute_reply.started":"2022-08-01T05:41:06.897955Z","shell.execute_reply":"2022-08-01T05:41:22.045099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Model Plot**","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.utils import plot_model\n\nplot_model(model)\nvalid_dataset","metadata":{"execution":{"iopub.status.busy":"2022-08-01T05:48:31.000951Z","iopub.execute_input":"2022-08-01T05:48:31.001778Z","iopub.status.idle":"2022-08-01T05:48:31.190599Z","shell.execute_reply.started":"2022-08-01T05:48:31.001737Z","shell.execute_reply":"2022-08-01T05:48:31.189388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_history = model.fit(\n    train_dataset,\n    steps_per_epoch=20,\n    validation_data=valid_dataset,\n    epochs=3\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T05:58:25.157395Z","iopub.execute_input":"2022-08-01T05:58:25.158503Z","iopub.status.idle":"2022-08-01T06:02:05.137373Z","shell.execute_reply.started":"2022-08-01T05:58:25.158453Z","shell.execute_reply":"2022-08-01T06:02:05.136441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📢 **Test Data - Prediction**","metadata":{}},{"cell_type":"code","source":"# Clean the text data\ntest_preprocessed = test['discourse_text'].apply(cleanup_text)\ntest_preprocessed\ntest['text_preprocessed'] =  test_preprocessed\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T06:06:24.753986Z","iopub.execute_input":"2022-08-01T06:06:24.754354Z","iopub.status.idle":"2022-08-01T06:06:24.770895Z","shell.execute_reply.started":"2022-08-01T06:06:24.754323Z","shell.execute_reply":"2022-08-01T06:06:24.769886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['processed'] = test.discourse_type + sep +test.discourse_text","metadata":{"execution":{"iopub.status.busy":"2022-08-01T06:06:30.986816Z","iopub.execute_input":"2022-08-01T06:06:30.987723Z","iopub.status.idle":"2022-08-01T06:06:30.993685Z","shell.execute_reply.started":"2022-08-01T06:06:30.987677Z","shell.execute_reply":"2022-08-01T06:06:30.992620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_processed = bert_encode(test.processed.astype(str), tokenizer)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T06:06:34.276971Z","iopub.execute_input":"2022-08-01T06:06:34.277605Z","iopub.status.idle":"2022-08-01T06:06:34.304209Z","shell.execute_reply.started":"2022-08-01T06:06:34.277570Z","shell.execute_reply":"2022-08-01T06:06:34.303315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📢 **Predict output**","metadata":{}},{"cell_type":"code","source":"preds = model.predict(test_processed, verbose=1)\npreds","metadata":{"execution":{"iopub.status.busy":"2022-08-01T06:06:38.078105Z","iopub.execute_input":"2022-08-01T06:06:38.078465Z","iopub.status.idle":"2022-08-01T06:06:41.049622Z","shell.execute_reply.started":"2022-08-01T06:06:38.078434Z","shell.execute_reply":"2022-08-01T06:06:41.048678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample['Ineffective'] = preds[:,0]\nsample['Adequate'] = preds[:,1]\nsample['Effective'] = preds[:,2]\nsample.sample(2)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T06:06:47.827864Z","iopub.execute_input":"2022-08-01T06:06:47.828721Z","iopub.status.idle":"2022-08-01T06:06:47.845578Z","shell.execute_reply.started":"2022-08-01T06:06:47.828674Z","shell.execute_reply":"2022-08-01T06:06:47.844613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📢 **Generate CSV File**","metadata":{}},{"cell_type":"code","source":"sample.to_csv(\"submission.csv\", index=False)\nprint('Successfully executed!')\nsample","metadata":{"execution":{"iopub.status.busy":"2022-08-01T06:07:09.524954Z","iopub.execute_input":"2022-08-01T06:07:09.525412Z","iopub.status.idle":"2022-08-01T06:07:09.553109Z","shell.execute_reply.started":"2022-08-01T06:07:09.525368Z","shell.execute_reply":"2022-08-01T06:07:09.552136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ⭐️⭐️**Thanks for visiting guys!**⭐️⭐️\n\n### ***Credits:***\n\n1. https://www.kaggle.com/code/imvision12/tensorflow-feedback-bert-baseline\n2. https://www.kaggle.com/code/venkatkumar001/nlpstarter-3-simple-ml-baseline-algo-with-kfold\n\n### ***If you know more about Huggingface transformer:***\n\n1. https://www.kaggle.com/code/venkatkumar001/nlp-starter2-hf-pretrain-finetune\n2. https://www.kaggle.com/code/venkatkumar001/transformeranatomy-encoder","metadata":{}}]}