{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"../input/feedback-prize-effectiveness/train.csv\")\ntrain[\"label\"] = train[\"discourse_effectiveness\"].replace({\"Ineffective\": 0, \"Adequate\": 1, \"Effective\": 2})\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:53:07.699166Z","iopub.execute_input":"2022-07-24T01:53:07.700675Z","iopub.status.idle":"2022-07-24T01:53:08.297294Z","shell.execute_reply.started":"2022-07-24T01:53:07.700632Z","shell.execute_reply":"2022-07-24T01:53:08.296258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom keras.layers import Dense, Input, Dropout,CuDNNLSTM\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model\nfrom transformers import TFBertModel\nimport transformers\n\n\n# Configuration\nBATCH_SIZE = 32\nMAX_LEN = 256 \nDROPOUT = 0.1 # 0.2\nLEARNING_RATE = 1e-5\nEPOCHS = 1#8\nAUTO = tf.data.experimental.AUTOTUNE\nMODEL = \"bert\" #\"bert\"\nMODEL_PATH = f\"../input/huggingface-bert-variants/{MODEL}-base-uncased/{MODEL}-base-uncased\"","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:53:08.298807Z","iopub.execute_input":"2022-07-24T01:53:08.299421Z","iopub.status.idle":"2022-07-24T01:53:08.308534Z","shell.execute_reply.started":"2022-07-24T01:53:08.299383Z","shell.execute_reply":"2022-07-24T01:53:08.307311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --upgrade -q wandb","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:53:08.311640Z","iopub.execute_input":"2022-07-24T01:53:08.312351Z","iopub.status.idle":"2022-07-24T01:55:46.290144Z","shell.execute_reply.started":"2022-07-24T01:53:08.312309Z","shell.execute_reply":"2022-07-24T01:55:46.289022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CONFIG = dict(competition = \"Feedback Prize Effectiveness\", \n              _wandb_kernel = \"iamleonie\",\n              dropout = DROPOUT,\n              learning_rate = LEARNING_RATE,\n              epochs = EPOCHS,\n              batch_size = BATCH_SIZE,\n              model = MODEL\n             )","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:55:46.293237Z","iopub.execute_input":"2022-07-24T01:55:46.293902Z","iopub.status.idle":"2022-07-24T01:55:46.299375Z","shell.execute_reply.started":"2022-07-24T01:55:46.293857Z","shell.execute_reply":"2022-07-24T01:55:46.298508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = transformers.BertTokenizer.from_pretrained('../input/huggingface-bert-variants/bert-base-uncased/bert-base-uncased')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:55:46.300926Z","iopub.execute_input":"2022-07-24T01:55:46.301578Z","iopub.status.idle":"2022-07-24T01:55:46.370876Z","shell.execute_reply.started":"2022-07-24T01:55:46.301543Z","shell.execute_reply":"2022-07-24T01:55:46.370018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sep = tokenizer.sep_token\nsep\ntrain['inputs'] = train.discourse_type + sep + train.discourse_text\ntrain.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:55:46.372323Z","iopub.execute_input":"2022-07-24T01:55:46.372671Z","iopub.status.idle":"2022-07-24T01:55:46.402259Z","shell.execute_reply.started":"2022-07-24T01:55:46.372636Z","shell.execute_reply":"2022-07-24T01:55:46.401210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Sample input sequence:')\nsample_sequence = train['inputs'].iloc[0]\nprint(sample_sequence)\n\nprint('\\nTokenized sequence:')\nprint(tokenizer.tokenize(sample_sequence))\n\ntoken = tokenizer(sample_sequence, \n                  max_length         = MAX_LEN, \n                  truncation         = True, \n                  padding            = 'max_length',\n                  add_special_tokens = True,\n                  return_tensors     = \"np\"\n                 )\n    \nprint('\\ninput_ids:')\nprint(token['input_ids'])\nprint('\\nattention_mask:')\nprint(token['attention_mask'])","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:55:46.403933Z","iopub.execute_input":"2022-07-24T01:55:46.404307Z","iopub.status.idle":"2022-07-24T01:55:46.423605Z","shell.execute_reply.started":"2022-07-24T01:55:46.404271Z","shell.execute_reply":"2022-07-24T01:55:46.422794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def bert_encode(texts, tokenizer, max_len = MAX_LEN):\n    input_ids = np.zeros((len(texts), max_len), dtype=\"int32\")\n    token_type_ids = np.zeros((len(texts), max_len), dtype=\"int32\")\n    attention_mask = np.zeros((len(texts), max_len), dtype=\"int32\")\n    \n    for i, text in enumerate(texts):\n        token = tokenizer(text, \n                          max_length         = max_len, \n                          truncation         = True, \n                          padding            = \"max_length\",\n                          add_special_tokens = True,\n                          return_tensors     = \"np\")\n        \n        input_ids[i] = token['input_ids']\n        token_type_ids[i] = token['token_type_ids']\n        attention_mask[i] = token['attention_mask']\n    return input_ids, attention_mask,token_type_ids, ","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:55:46.424820Z","iopub.execute_input":"2022-07-24T01:55:46.425252Z","iopub.status.idle":"2022-07-24T01:55:46.433837Z","shell.execute_reply.started":"2022-07-24T01:55:46.425217Z","shell.execute_reply":"2022-07-24T01:55:46.431802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_valid, y_train, y_valid = train_test_split(train['inputs'], train['label'], test_size=0.12, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:55:46.438664Z","iopub.execute_input":"2022-07-24T01:55:46.439641Z","iopub.status.idle":"2022-07-24T01:55:46.456546Z","shell.execute_reply.started":"2022-07-24T01:55:46.439606Z","shell.execute_reply":"2022-07-24T01:55:46.455665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = bert_encode(X_train.astype(str), tokenizer)\nX_valid = bert_encode(X_valid.astype(str), tokenizer)\n\ny_train = y_train.values\ny_valid = y_valid.values","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:55:46.457960Z","iopub.execute_input":"2022-07-24T01:55:46.458455Z","iopub.status.idle":"2022-07-24T01:56:53.489890Z","shell.execute_reply.started":"2022-07-24T01:55:46.458419Z","shell.execute_reply":"2022-07-24T01:56:53.488919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((X_train, y_train))\n    .repeat()\n    .shuffle(1024)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((X_valid, y_valid))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:56:53.491341Z","iopub.execute_input":"2022-07-24T01:56:53.491689Z","iopub.status.idle":"2022-07-24T01:56:53.511197Z","shell.execute_reply.started":"2022-07-24T01:56:53.491654Z","shell.execute_reply":"2022-07-24T01:56:53.510183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model(bert_model, max_len=MAX_LEN):    \n    input_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_ids\")\n    token_type_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"token_type_ids\")\n    attention_mask = Input(shape=(max_len,), dtype=tf.int32, name=\"attention_mask\")\n\n    sequence_output = bert_model(input_ids, token_type_ids = token_type_ids, attention_mask=attention_mask)[0]\n    clf_output = sequence_output[:, 0, :]\n    out = Dropout(.1)(clf_output)\n    #out = CuDNNLSTM(100)(clf_output)\n    out = Dense(512,activation='tanh')(out)\n    out = Dropout(0.1)(out)\n    out = Dense(512,activation='tanh')(out)\n    out = Dropout(0.1)(out)\n    out = Dense(3, activation='softmax')(out)\n    optimizer = Adam(lr = 1e-5, epsilon = 1e-08,clipvalue = 10)\n    model = Model(inputs=[input_ids, token_type_ids, attention_mask], outputs=out)\n    model.compile(optimizer=optimizer, loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:56:53.512712Z","iopub.execute_input":"2022-07-24T01:56:53.513512Z","iopub.status.idle":"2022-07-24T01:56:53.523171Z","shell.execute_reply.started":"2022-07-24T01:56:53.513474Z","shell.execute_reply":"2022-07-24T01:56:53.522166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transformer_layer = (TFBertModel.from_pretrained('../input/huggingface-bert-variants/bert-base-uncased/bert-base-uncased'))\nmodel = build_model(transformer_layer, max_len=MAX_LEN)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:56:53.524724Z","iopub.execute_input":"2022-07-24T01:56:53.525391Z","iopub.status.idle":"2022-07-24T01:57:00.212257Z","shell.execute_reply.started":"2022-07-24T01:56:53.525355Z","shell.execute_reply":"2022-07-24T01:57:00.211298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import keras\ncallbacks = [\n        keras.callbacks.ModelCheckpoint(\n        \"best_model.h5\", save_best_only=True, monitor=\"val_loss\", mode = 'min', verbose = 1,\n        ),\n    ]","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-07-24T01:57:00.213905Z","iopub.execute_input":"2022-07-24T01:57:00.214560Z","iopub.status.idle":"2022-07-24T01:57:00.220131Z","shell.execute_reply.started":"2022-07-24T01:57:00.214520Z","shell.execute_reply":"2022-07-24T01:57:00.219142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_history = model.fit(\n    train_dataset,\n    steps_per_epoch=200,\n    validation_data=valid_dataset,\n    callbacks=callbacks,\n    epochs=40)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:57:00.221563Z","iopub.execute_input":"2022-07-24T01:57:00.221909Z","iopub.status.idle":"2022-07-24T03:17:19.472182Z","shell.execute_reply.started":"2022-07-24T01:57:00.221874Z","shell.execute_reply":"2022-07-24T03:17:19.470763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv(\"../input/feedback-prize-effectiveness/test.csv\")\ntest['text'] = test.discourse_type + sep +test.discourse_text\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T03:17:19.473305Z","iopub.status.idle":"2022-07-24T03:17:19.475184Z","shell.execute_reply.started":"2022-07-24T03:17:19.474897Z","shell.execute_reply":"2022-07-24T03:17:19.474924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_text = bert_encode(test.text.astype(str), tokenizer)\nsub = pd.read_csv(\"../input/feedback-prize-effectiveness/sample_submission.csv\")\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T03:17:19.476785Z","iopub.status.idle":"2022-07-24T03:17:19.477573Z","shell.execute_reply.started":"2022-07-24T03:17:19.477301Z","shell.execute_reply":"2022-07-24T03:17:19.477325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = keras.models.load_model(\"/kaggle/working/best_model.h5\",custom_objects={'TFBertModel': TFBertModel})\npreds = model.predict(test_text, verbose=1)\npreds","metadata":{"execution":{"iopub.status.busy":"2022-07-24T03:17:19.479084Z","iopub.status.idle":"2022-07-24T03:17:19.479891Z","shell.execute_reply.started":"2022-07-24T03:17:19.479595Z","shell.execute_reply":"2022-07-24T03:17:19.479647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub['Ineffective'] = preds[:,0]\nsub['Adequate'] = preds[:,1]\nsub['Effective'] = preds[:,2]\nsub","metadata":{"execution":{"iopub.status.busy":"2022-07-24T03:17:19.481305Z","iopub.status.idle":"2022-07-24T03:17:19.482071Z","shell.execute_reply.started":"2022-07-24T03:17:19.481826Z","shell.execute_reply":"2022-07-24T03:17:19.481849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T03:17:19.483476Z","iopub.status.idle":"2022-07-24T03:17:19.484275Z","shell.execute_reply.started":"2022-07-24T03:17:19.484029Z","shell.execute_reply":"2022-07-24T03:17:19.484053Z"},"trusted":true},"execution_count":null,"outputs":[]}]}