{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nfrom transformers import BertTokenizer\nimport tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2022-08-05T02:52:04.384674Z","iopub.execute_input":"2022-08-05T02:52:04.385219Z","iopub.status.idle":"2022-08-05T02:52:10.579038Z","shell.execute_reply.started":"2022-08-05T02:52:04.385090Z","shell.execute_reply":"2022-08-05T02:52:10.577587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv_path = \"../input/feedback-prize-effectiveness/train.csv\"\ntrainDf = pd.read_csv(train_csv_path)\ntrainDf.drop([\"discourse_id\"], axis=1, inplace=True)\n\ntype2value = {k:v+1 for v, k in enumerate(trainDf['discourse_type'].unique())}\neffectiveness2target = {k:v+1 for v, k in enumerate(trainDf['discourse_effectiveness'].unique())}\ntarget2effectiveness = {v+1:k for v, k in enumerate(trainDf['discourse_effectiveness'].unique())}\n\n\ntrainDf['discourse_effectiveness'].replace(effectiveness2target, inplace=True)\ntrainDf['discourse_type'].replace(type2value, inplace=True)\n\ntrainDf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T02:52:10.580914Z","iopub.execute_input":"2022-08-05T02:52:10.581559Z","iopub.status.idle":"2022-08-05T02:52:10.983405Z","shell.execute_reply.started":"2022-08-05T02:52:10.581520Z","shell.execute_reply":"2022-08-05T02:52:10.981930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **GroupBy Essay Id**","metadata":{}},{"cell_type":"code","source":"groupedTrainDf = trainDf.groupby(\"essay_id\")\ngroupedTrainDf.get_group(\"007ACE74B050\")","metadata":{"execution":{"iopub.status.busy":"2022-08-05T02:52:10.984936Z","iopub.execute_input":"2022-08-05T02:52:10.985420Z","iopub.status.idle":"2022-08-05T02:52:11.015421Z","shell.execute_reply.started":"2022-08-05T02:52:10.985373Z","shell.execute_reply":"2022-08-05T02:52:11.014504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **DATA GENERATOR**","metadata":{}},{"cell_type":"markdown","source":"The data is generated by custom generator. The generated data is in a format of each **essay is considered as sequences of sentences with each sentence map to representive type**. The dataset generated can be used to predict the **target as a sequence** for the given **sequence of sentences and its type**.","metadata":{}},{"cell_type":"markdown","source":"* The sentence sequences is padded to maximum size of 23.\n* The shape of sequence will be (batch_size, 23, 1, 512)\n* The sentence type and target is padded to the maximum limit\n* The BertTokenizer is used to tokenize the texts","metadata":{}},{"cell_type":"code","source":"class BertDataGen(tf.keras.utils.Sequence):\n    def __init__(self, df, shuffle=True, training=True, batchSize = 8, Npadding=23):\n        self.df = df\n        self.shuffle = shuffle\n        self.training = training\n        self.batchSize = batchSize\n        self.essayIds = df.essay_id.unique()\n        self.groupedDf = df.groupby('essay_id')\n        self.N = len(self.groupedDf.first())\n        self.tokenizer = BertTokenizer.from_pretrained(\"bert-base-uncased\")\n        self.Npadding = Npadding\n        \n\n    def on_epoch_end(self):\n        if self.shuffle:\n            self.essayIds = np.random.shuffle(self.essayIds)            \n            \n            \n    def __getitem__(self, index):\n        essayIdsBatch = self.essayIds[index*self.batchSize : (index + 1)*self.batchSize]\n        batchInputIdsTensorArray = []\n        batchTokenTypeIdsTensorArray = []\n        batchAttentationMaskArray = []\n\n        batchTypeTensorArray = []\n        batchTargetTensorArray = []\n\n        for i, essayId in enumerate(essayIdsBatch):\n            group = self.groupedDf.get_group(essayId)\n            Ginput_input_ids, Gtoken_type_ids, Gattention_mask = self.__encodeText(group['discourse_text'].values)\n            discourseTypeGroup = group['discourse_type'].values\n            targetGroup = group['discourse_effectiveness'].values\n            \n            discourseTypeGroup = tf.pad(discourseTypeGroup, paddings=[[0, self.Npadding - discourseTypeGroup.shape[0]]])\n            targetGroup = tf.pad(targetGroup, paddings=[[0, self.Npadding - targetGroup.shape[0]]])\n            \n            batchInputIdsTensorArray.append(Ginput_input_ids)\n            batchTokenTypeIdsTensorArray.append(Gtoken_type_ids)\n            batchAttentationMaskArray.append(Gattention_mask)\n\n            batchTypeTensorArray.append(discourseTypeGroup)\n            batchTargetTensorArray.append(targetGroup)\n            \n        batchInputIdsTensorArray = tf.convert_to_tensor(batchInputIdsTensorArray)\n        batchTokenTypeIdsTensorArray = tf.convert_to_tensor(batchTokenTypeIdsTensorArray)\n        batchAttentationMaskArray = tf.convert_to_tensor(batchAttentationMaskArray)\n        batchTypeTensorArray = tf.convert_to_tensor(batchTypeTensorArray)\n        batchTargetTensorArray = tf.convert_to_tensor(batchTargetTensorArray)\n        \n        batchTypeTensorArray = tf.reshape(batchTypeTensorArray, [-1, self.Npadding, 1])\n        \n        return ((batchInputIdsTensorArray ,batchTokenTypeIdsTensorArray, batchAttentationMaskArray), batchTypeTensorArray), batchTargetTensorArray\n\n    \n    def __len__(self):\n        return self.N // self.batchSize\n    \n    def __encodeText(self, texts):\n        input_ids = []\n        token_type_ids = []\n        attention_mask = []\n        for text in texts:\n            token = self.tokenizer(text, max_length=512, padding='max_length', return_tensors=\"tf\")\n            input_ids.append(token['input_ids'])\n            token_type_ids.append(token['token_type_ids'])\n            attention_mask.append(token['attention_mask'])   \n        input_ids = np.array(input_ids)\n        token_type_ids = np.array(token_type_ids)\n        attention_mask = np.array(attention_mask)\n        \n        input_ids = tf.pad(input_ids, paddings=[[0, 23-input_ids.shape[0]],[0, 0],[0, 0]])\n        token_type_ids = tf.pad(token_type_ids, paddings=[[0, self.Npadding-token_type_ids.shape[0]],[0, 0],[0, 0]])\n        attention_mask = tf.pad(attention_mask, paddings=[[0, self.Npadding-attention_mask.shape[0]],[0, 0],[0, 0]])\n        return (input_ids, token_type_ids, attention_mask)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T02:52:11.017575Z","iopub.execute_input":"2022-08-05T02:52:11.018159Z","iopub.status.idle":"2022-08-05T02:52:12.285687Z","shell.execute_reply.started":"2022-08-05T02:52:11.018103Z","shell.execute_reply":"2022-08-05T02:52:12.284417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainGen = BertDataGen(trainDf, Npadding=trainDf.essay_id.value_counts()[0])","metadata":{"execution":{"iopub.status.busy":"2022-08-05T02:52:12.287265Z","iopub.execute_input":"2022-08-05T02:52:12.287626Z","iopub.status.idle":"2022-08-05T02:52:21.069371Z","shell.execute_reply.started":"2022-08-05T02:52:12.287595Z","shell.execute_reply":"2022-08-05T02:52:21.068228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"((batchInputIdsTensorArray ,batchTokenTypeIdsTensorArray, batchAttentationMaskArray), batchTypeTensorArray), batchTargetTensorArray = trainGen.__getitem__(1)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T02:52:21.070585Z","iopub.execute_input":"2022-08-05T02:52:21.070902Z","iopub.status.idle":"2022-08-05T02:52:21.277404Z","shell.execute_reply.started":"2022-08-05T02:52:21.070872Z","shell.execute_reply":"2022-08-05T02:52:21.275097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Shapes of generated data","metadata":{}},{"cell_type":"code","source":"((batchInputIdsTensorArray.shape ,batchTokenTypeIdsTensorArray.shape, batchAttentationMaskArray.shape), batchTypeTensorArray.shape), batchTargetTensorArray.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-05T02:52:21.278754Z","iopub.execute_input":"2022-08-05T02:52:21.279197Z","iopub.status.idle":"2022-08-05T02:52:21.285946Z","shell.execute_reply.started":"2022-08-05T02:52:21.279154Z","shell.execute_reply":"2022-08-05T02:52:21.285022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# These are the bert inputs.argpartition# (batchInputIdsTensorArray ,batchTokenTypeIdsTensorArray, batchAttentationMaskArray)\n\n# batchTypeTensorArray is a sequences represent a type of sentence in an essay.\n# batchTargetTensorArray is a target for the sequences of sentences in an essay.","metadata":{"execution":{"iopub.status.busy":"2022-08-05T02:52:21.287471Z","iopub.execute_input":"2022-08-05T02:52:21.288089Z","iopub.status.idle":"2022-08-05T02:52:21.296804Z","shell.execute_reply.started":"2022-08-05T02:52:21.288057Z","shell.execute_reply":"2022-08-05T02:52:21.295906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Will this work..?**","metadata":{}}]}