{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-11T01:28:05.508712Z","iopub.execute_input":"2022-08-11T01:28:05.509650Z","iopub.status.idle":"2022-08-11T01:28:05.531304Z","shell.execute_reply.started":"2022-08-11T01:28:05.509529Z","shell.execute_reply":"2022-08-11T01:28:05.530283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport random\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Dense, Input, Dropout\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model\nimport transformers\nfrom transformers import BertTokenizer,TFBertModel\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nos.getcwd()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T01:28:05.533691Z","iopub.execute_input":"2022-08-11T01:28:05.534394Z","iopub.status.idle":"2022-08-11T01:28:21.569673Z","shell.execute_reply.started":"2022-08-11T01:28:05.534357Z","shell.execute_reply":"2022-08-11T01:28:21.568763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = \"../input/feedback-prize-effectiveness\"\ntrain_text_path = os.path.join(path,\"train\")\ntrain_text_path = os.path.join(path,\"test\")\ntrain = pd.read_csv(\"../input/feedback-prize-effectiveness/train.csv\")\ntest = pd.read_csv(\"../input/feedback-prize-effectiveness/test.csv\")\nsample_submission = pd.read_csv(\"../input/feedback-prize-effectiveness/sample_submission.csv\")\ndisplay(train.head())\ndisplay(test.head())\ndisplay(sample_submission.head())","metadata":{"execution":{"iopub.status.busy":"2022-08-11T01:28:21.571091Z","iopub.execute_input":"2022-08-11T01:28:21.572535Z","iopub.status.idle":"2022-08-11T01:28:21.892948Z","shell.execute_reply.started":"2022-08-11T01:28:21.572497Z","shell.execute_reply":"2022-08-11T01:28:21.892026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_dict = {\"Effective\" : 2, \"Adequate\" : 1, \"Ineffective\" : 0}\ntrain[\"label\"] = train[\"discourse_effectiveness\"].replace(target_dict)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T01:28:21.896029Z","iopub.execute_input":"2022-08-11T01:28:21.896381Z","iopub.status.idle":"2022-08-11T01:28:21.921118Z","shell.execute_reply.started":"2022-08-11T01:28:21.896354Z","shell.execute_reply":"2022-08-11T01:28:21.920063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\nEPOCHS = 3\nBATCH_SIZE = 8\nMAX_LENGTH = 400","metadata":{"execution":{"iopub.status.busy":"2022-08-11T01:28:21.922732Z","iopub.execute_input":"2022-08-11T01:28:21.923143Z","iopub.status.idle":"2022-08-11T01:28:21.929591Z","shell.execute_reply.started":"2022-08-11T01:28:21.923106Z","shell.execute_reply":"2022-08-11T01:28:21.928434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def encode(texts, tokenizer, max_len = MAX_LENGTH):\n  input_ids = []\n  token_type_ids = []\n  attention_mask = []\n  \n  for text in texts:\n    token = tokenizer(text, max_length = max_len,\n                      truncation = True,\n                      padding = \"max_length\",\n                      add_special_tokens = True\n               )\n    input_ids.append(token[\"input_ids\"])\n    token_type_ids.append(token[\"token_type_ids\"])\n    attention_mask.append(token[\"attention_mask\"])\n  return np.array(input_ids), np.array(token_type_ids), np.array(attention_mask)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T01:28:21.931369Z","iopub.execute_input":"2022-08-11T01:28:21.932222Z","iopub.status.idle":"2022-08-11T01:28:21.940070Z","shell.execute_reply.started":"2022-08-11T01:28:21.932183Z","shell.execute_reply":"2022-08-11T01:28:21.938831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tokenizer = BertTokenizer.from_pretrained('bert-base-cased')\ntokenizer = BertTokenizer.from_pretrained('/kaggle/input/bert-based-cased')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T01:28:21.941699Z","iopub.execute_input":"2022-08-11T01:28:21.942538Z","iopub.status.idle":"2022-08-11T01:28:22.004783Z","shell.execute_reply.started":"2022-08-11T01:28:21.942503Z","shell.execute_reply":"2022-08-11T01:28:22.003822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sep = tokenizer.sep_token\ndisplay(sep)\ncls = tokenizer.cls_token\ndisplay(cls)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T01:28:22.006330Z","iopub.execute_input":"2022-08-11T01:28:22.006677Z","iopub.status.idle":"2022-08-11T01:28:22.017659Z","shell.execute_reply.started":"2022-08-11T01:28:22.006644Z","shell.execute_reply":"2022-08-11T01:28:22.016446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"inputs\"] = cls+train[\"discourse_type\"]+sep+train[\"discourse_text\"]\ndisplay(train.head(3))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T01:28:22.019664Z","iopub.execute_input":"2022-08-11T01:28:22.020529Z","iopub.status.idle":"2022-08-11T01:28:22.053954Z","shell.execute_reply.started":"2022-08-11T01:28:22.020494Z","shell.execute_reply":"2022-08-11T01:28:22.053010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_valid, y_train, y_valid = train_test_split(train[\"inputs\"],\n                                train[\"label\"],\n                                test_size = 0.2,\n                                random_state = 42)\nX_train","metadata":{"execution":{"iopub.status.busy":"2022-08-11T01:28:22.057541Z","iopub.execute_input":"2022-08-11T01:28:22.058555Z","iopub.status.idle":"2022-08-11T01:28:22.074359Z","shell.execute_reply.started":"2022-08-11T01:28:22.058513Z","shell.execute_reply":"2022-08-11T01:28:22.073376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_token = encode(X_train.astype(str), tokenizer)\nX_valid_token = encode(X_valid.astype(str), tokenizer)\ny_train_token = y_train.values\ny_valid_token = y_valid.values","metadata":{"execution":{"iopub.status.busy":"2022-08-11T01:28:22.075881Z","iopub.execute_input":"2022-08-11T01:28:22.076596Z","iopub.status.idle":"2022-08-11T01:29:13.535632Z","shell.execute_reply.started":"2022-08-11T01:28:22.076558Z","shell.execute_reply":"2022-08-11T01:29:13.534634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X_train_token[0].shape, X_train_token[1].shape, X_train_token[2].shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T01:29:13.536944Z","iopub.execute_input":"2022-08-11T01:29:13.537911Z","iopub.status.idle":"2022-08-11T01:29:13.544223Z","shell.execute_reply.started":"2022-08-11T01:29:13.537874Z","shell.execute_reply":"2022-08-11T01:29:13.542842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((X_train_token, y_train_token))\n    .repeat()\n    .shuffle(2048)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((X_valid_token, y_valid_token))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T01:29:13.545879Z","iopub.execute_input":"2022-08-11T01:29:13.546278Z","iopub.status.idle":"2022-08-11T01:29:14.188925Z","shell.execute_reply.started":"2022-08-11T01:29:13.546243Z","shell.execute_reply":"2022-08-11T01:29:14.187930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model(pretrained_model, max_len=MAX_LENGTH):\n  input_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_ids\")\n  token_type_ids = Input(shape = (max_len,), dtype = tf.int32, name=\"token_type_ids\")\n  attention_mask = Input(shape = (max_len,), dtype = tf.int32, name=\"attention_mask\")\n\n  # fine_tune\n  sequence_output = pretrained_model(input_ids,token_type_ids=token_type_ids,\n                      attention_mask = attention_mask)[0]\n  clf_output = sequence_output[:, 0, :]\n  clf_output = Dropout(0.1)(clf_output)\n  out = Dense(3,activation = \"softmax\")(clf_output)\n\n  # Model_compiling\n  model = Model(inputs = [input_ids, token_type_ids, attention_mask],\n                outputs = out)\n  model.compile(Adam(lr = 1e-5), loss = \"sparse_categorical_crossentropy\",\n                metrics = [\"accuracy\"])\n  return model","metadata":{"execution":{"iopub.status.busy":"2022-08-11T01:29:14.190233Z","iopub.execute_input":"2022-08-11T01:29:14.190580Z","iopub.status.idle":"2022-08-11T01:29:14.199303Z","shell.execute_reply.started":"2022-08-11T01:29:14.190547Z","shell.execute_reply":"2022-08-11T01:29:14.198252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# transformers_layer = TFBertModel.from_pretrained('bert-base-cased')\ntransformers_layer = TFBertModel.from_pretrained('/kaggle/input/bert-based-cased-model')\nmodel = build_model(transformers_layer, max_len=MAX_LENGTH)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T01:29:14.201120Z","iopub.execute_input":"2022-08-11T01:29:14.201791Z","iopub.status.idle":"2022-08-11T01:29:28.089594Z","shell.execute_reply.started":"2022-08-11T01:29:14.201739Z","shell.execute_reply":"2022-08-11T01:29:28.088609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_history = model.fit(\n    train_dataset,\n    steps_per_epoch = 4000,\n    validation_data = valid_dataset,\n    epochs = 2\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T01:29:28.090937Z","iopub.execute_input":"2022-08-11T01:29:28.091891Z","iopub.status.idle":"2022-08-11T03:07:19.058368Z","shell.execute_reply.started":"2022-08-11T01:29:28.091848Z","shell.execute_reply":"2022-08-11T03:07:19.057349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntest[\"Input\"] = cls + test[\"discourse_type\"] + sep + test[\"discourse_text\"]\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:07:19.061029Z","iopub.execute_input":"2022-08-11T03:07:19.061781Z","iopub.status.idle":"2022-08-11T03:07:19.077000Z","shell.execute_reply.started":"2022-08-11T03:07:19.061725Z","shell.execute_reply":"2022-08-11T03:07:19.075996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test_token = encode(test[\"Input\"].astype(str), tokenizer)\npreds = model.predict(X_test_token, verbose = 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:07:19.078714Z","iopub.execute_input":"2022-08-11T03:07:19.079448Z","iopub.status.idle":"2022-08-11T03:07:22.128476Z","shell.execute_reply.started":"2022-08-11T03:07:19.079411Z","shell.execute_reply":"2022-08-11T03:07:22.127544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = sample_submission.copy()\nsubmission[\"Ineffective\"] = preds[:,0]\nsubmission[\"Adequate\"] = preds[:,1]\nsubmission[\"Effective\"] = preds[:,2]\nsubmission.to_csv(\"submission.csv\",index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:07:22.131675Z","iopub.execute_input":"2022-08-11T03:07:22.132006Z","iopub.status.idle":"2022-08-11T03:07:22.152503Z","shell.execute_reply.started":"2022-08-11T03:07:22.131978Z","shell.execute_reply":"2022-08-11T03:07:22.151604Z"},"trusted":true},"execution_count":null,"outputs":[]}]}