{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# import the libraries \nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras import layers\nfrom tensorflow import keras\nfrom tensorflow.keras.optimizers import Adam\nimport transformers\nimport tqdm\nfrom sklearn import preprocessing\nfrom sklearn.model_selection import train_test_split\n","metadata":{"execution":{"iopub.status.busy":"2022-08-08T13:42:08.766893Z","iopub.execute_input":"2022-08-08T13:42:08.767537Z","iopub.status.idle":"2022-08-08T13:42:15.122100Z","shell.execute_reply.started":"2022-08-08T13:42:08.767451Z","shell.execute_reply":"2022-08-08T13:42:15.121041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## The dataset Overview\n* Lead - an introduction that begins with a statistic, a quotation, a description, or some other device to grab the reader’s attention and point toward the thesis\n* Position - an opinion or conclusion on the main question\n* Claim - a claim that supports the position\n* Counterclaim - a claim that refutes another claim or gives an opposing reason to the position\n* Rebuttal - a claim that refutes a counterclaim\n* Evidence - ideas or examples that support claims, counterclaims, or rebuttals.\n\nYour task is to predict the quality rating of each discourse element. Human readers rated each rhetorical or argumentative element, in order of increasing quality, as one of.\n\n        *     Ineffective\n        *     Adequate\n        *     Effective","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/feedback-prize-effectiveness/train.csv')\ntest_df  = pd.read_csv('/kaggle/input/feedback-prize-effectiveness/test.csv')\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T13:42:15.125765Z","iopub.execute_input":"2022-08-08T13:42:15.126999Z","iopub.status.idle":"2022-08-08T13:42:15.416883Z","shell.execute_reply.started":"2022-08-08T13:42:15.126961Z","shell.execute_reply":"2022-08-08T13:42:15.415921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Map the labels effectivenes category ","metadata":{}},{"cell_type":"code","source":"target_column = \"discourse_effectiveness\"\nle = preprocessing.LabelEncoder()\nle.fit(train_df[target_column])\ntrain_df['target'] = le.transform(train_df[target_column])","metadata":{"execution":{"iopub.status.busy":"2022-08-08T13:52:19.936150Z","iopub.execute_input":"2022-08-08T13:52:19.936533Z","iopub.status.idle":"2022-08-08T13:52:19.952802Z","shell.execute_reply.started":"2022-08-08T13:52:19.936500Z","shell.execute_reply":"2022-08-08T13:52:19.951780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##  Text Preprocessing ","metadata":{}},{"cell_type":"code","source":"train_df['text'] = train_df['discourse_text']","metadata":{"execution":{"iopub.status.busy":"2022-08-08T13:53:54.443240Z","iopub.execute_input":"2022-08-08T13:53:54.443666Z","iopub.status.idle":"2022-08-08T13:53:54.451803Z","shell.execute_reply.started":"2022-08-08T13:53:54.443633Z","shell.execute_reply":"2022-08-08T13:53:54.450626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import nltk ,re\nfrom nltk.corpus import stopwords\nfrom nltk.stem.porter import PorterStemmer\nimport string \nfrom nltk.stem import WordNetLemmatizer\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2022-08-08T13:53:57.301916Z","iopub.execute_input":"2022-08-08T13:53:57.303078Z","iopub.status.idle":"2022-08-08T13:53:57.308572Z","shell.execute_reply.started":"2022-08-08T13:53:57.303031Z","shell.execute_reply":"2022-08-08T13:53:57.307602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#defining the object for Lemmatization\nnltk.download('wordnet')\nnltk.download('stopwords')","metadata":{"execution":{"iopub.status.busy":"2022-08-08T13:53:57.589377Z","iopub.execute_input":"2022-08-08T13:53:57.590067Z","iopub.status.idle":"2022-08-08T13:53:57.597214Z","shell.execute_reply.started":"2022-08-08T13:53:57.590029Z","shell.execute_reply":"2022-08-08T13:53:57.596194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wordnet_lemmatizer = WordNetLemmatizer()\nstopwords=stopwords.words('english')\nstemmer=PorterStemmer()\n# clean unwanted text like stopwords, @(Mention), https(url), #(Hashtag), punctuations\ndef removeUnwantedText(text):\n    #remove urls\n    if text == np.NaN or type(text) != str:\n      text = \" \"\n    text = re.sub(r'http\\S+', \" \", text)\n    text = re.sub(r'@\\w+',' ',text)\n    text = re.sub(r'#\\w+', ' ', text)\n    text = re.sub('r<.*?>',' ', text)\n    # html tags\n    text = text.lower()\n    text = text.split()\n    text = \" \".join([word for word in text if not word in stopwords])\n    for punctuation in string.punctuation:\n        text = text.replace(punctuation, \"\")\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-08-08T13:53:57.869686Z","iopub.execute_input":"2022-08-08T13:53:57.870247Z","iopub.status.idle":"2022-08-08T13:53:57.878948Z","shell.execute_reply.started":"2022-08-08T13:53:57.870215Z","shell.execute_reply":"2022-08-08T13:53:57.877999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab_size = 16000  \nmaxlen = 64  # Only consider the first 200 words of each movie review","metadata":{"execution":{"iopub.status.busy":"2022-08-08T13:53:58.621459Z","iopub.execute_input":"2022-08-08T13:53:58.621927Z","iopub.status.idle":"2022-08-08T13:53:58.627000Z","shell.execute_reply.started":"2022-08-08T13:53:58.621888Z","shell.execute_reply":"2022-08-08T13:53:58.625977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing import text\nfrom tensorflow.keras.preprocessing import sequence","metadata":{"execution":{"iopub.status.busy":"2022-08-08T13:54:01.014594Z","iopub.execute_input":"2022-08-08T13:54:01.015276Z","iopub.status.idle":"2022-08-08T13:54:01.020228Z","shell.execute_reply.started":"2022-08-08T13:54:01.015239Z","shell.execute_reply":"2022-08-08T13:54:01.019098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = text.Tokenizer(num_words=vocab_size)\ntokenizer.fit_on_texts(train_df[\"text\"])\ndef prep_text(texts, tokenizer, max_sequence_length):\n    # Turns text into into padded sequences.\n    text_sequences = tokenizer.texts_to_sequences(texts)\n    return sequence.pad_sequences(text_sequences, maxlen=max_sequence_length)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T13:54:01.604125Z","iopub.execute_input":"2022-08-08T13:54:01.605188Z","iopub.status.idle":"2022-08-08T13:54:03.026374Z","shell.execute_reply.started":"2022-08-08T13:54:01.605138Z","shell.execute_reply":"2022-08-08T13:54:03.025234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train validation split","metadata":{}},{"cell_type":"code","source":"## coverting to matrix \nx= prep_text(train_df['text'],tokenizer,maxlen)\nx= np.array(x)\ny =np.array(train_df['target'])\nx_train, x_val, y_train, y_val = train_test_split(x, y, test_size=0.20, random_state=4)\ny_val = tf.one_hot(y_val, 3)\ny_train= tf.one_hot(y_train, 3)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T13:57:37.737212Z","iopub.execute_input":"2022-08-08T13:57:37.737593Z","iopub.status.idle":"2022-08-08T13:57:39.273931Z","shell.execute_reply.started":"2022-08-08T13:57:37.737559Z","shell.execute_reply":"2022-08-08T13:57:39.272935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Build Transformer Model ","metadata":{}},{"cell_type":"code","source":"class TransformerBlock(layers.Layer):\n    def __init__(self, embed_dim, num_heads, ff_dim, rate=0.1):\n        super(TransformerBlock, self).__init__()\n        self.att = layers.MultiHeadAttention(num_heads=num_heads, key_dim=embed_dim)\n        self.ffn = keras.Sequential(\n            [layers.Dense(ff_dim, activation=\"relu\"), layers.Dense(embed_dim),]\n        )\n        self.layernorm1 = layers.LayerNormalization(epsilon=1e-6)\n        self.layernorm2 = layers.LayerNormalization(epsilon=1e-6)\n        self.dropout1 = layers.Dropout(rate)\n        self.dropout2 = layers.Dropout(rate)\n\n    def call(self, inputs, training):\n        attn_output = self.att(inputs, inputs)\n        attn_output = self.dropout1(attn_output, training=training)\n        out1 = self.layernorm1(inputs + attn_output)\n        ffn_output = self.ffn(out1)\n        ffn_output = self.dropout2(ffn_output, training=training)\n        return self.layernorm2(out1 + ffn_output)\n\nclass TokenAndPositionEmbedding(layers.Layer):\n    def __init__(self, maxlen, vocab_size, embed_dim):\n        super(TokenAndPositionEmbedding, self).__init__()\n        self.token_emb = layers.Embedding(input_dim=vocab_size, output_dim=embed_dim)\n        self.pos_emb = layers.Embedding(input_dim=maxlen, output_dim=embed_dim)\n\n    def call(self, x):\n        maxlen = tf.shape(x)[-1]\n        positions = tf.range(start=0, limit=maxlen, delta=1)\n        positions = self.pos_emb(positions)\n        x = self.token_emb(x)\n        return x + positions","metadata":{"execution":{"iopub.status.busy":"2022-08-08T13:57:41.340167Z","iopub.execute_input":"2022-08-08T13:57:41.340527Z","iopub.status.idle":"2022-08-08T13:57:41.351653Z","shell.execute_reply.started":"2022-08-08T13:57:41.340495Z","shell.execute_reply":"2022-08-08T13:57:41.350615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embed_dim = 64  # Embedding size for each token\nnum_heads = 4  # Number of attention heads\nff_dim = 64  # Hidden layer size in feed forward network inside transformer\n\ninputs = layers.Input(shape=(maxlen,))\nembedding_layer = TokenAndPositionEmbedding(maxlen, vocab_size, embed_dim)\nx = embedding_layer(inputs)\ntransformer_block = TransformerBlock(embed_dim, num_heads, ff_dim)\nx = transformer_block(x)\nx = layers.GlobalAveragePooling1D()(x)\nx = layers.Dropout(0.1)(x)\nx = layers.Dense(64, activation=\"relu\")(x)\nx = layers.Dropout(0.1)(x)\noutputs = layers.Dense(3, activation=\"softmax\")(x)\n\nmodel = keras.Model(inputs=inputs, outputs=outputs)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T13:57:41.498120Z","iopub.execute_input":"2022-08-08T13:57:41.499849Z","iopub.status.idle":"2022-08-08T13:57:41.653506Z","shell.execute_reply.started":"2022-08-08T13:57:41.499795Z","shell.execute_reply":"2022-08-08T13:57:41.652484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train and Validate ","metadata":{}},{"cell_type":"code","source":"model.compile(\"adam\", \"categorical_crossentropy\", metrics=[\"accuracy\"])\nhistory = model.fit(\n    x_train, y_train, batch_size=32, epochs=5, validation_data=(x_val, y_val)\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T13:58:09.122920Z","iopub.execute_input":"2022-08-08T13:58:09.123884Z","iopub.status.idle":"2022-08-08T13:58:43.180584Z","shell.execute_reply.started":"2022-08-08T13:58:09.123848Z","shell.execute_reply":"2022-08-08T13:58:43.179609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission Section \n","metadata":{}},{"cell_type":"code","source":"x_test = prep_text(test_df['discourse_type'],tokenizer,maxlen)\nx_test= np.array(x_test)\ntest_df[target_column] = le.inverse_transform(tf.argmax(model.predict(x_test), axis = 1).numpy())\ntest_df.to_csv(\"submission.csv\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-08T14:00:24.586481Z","iopub.execute_input":"2022-08-08T14:00:24.586997Z","iopub.status.idle":"2022-08-08T14:00:24.592795Z","shell.execute_reply.started":"2022-08-08T14:00:24.586966Z","shell.execute_reply":"2022-08-08T14:00:24.591850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-08-08T14:03:00.311829Z","iopub.execute_input":"2022-08-08T14:03:00.312248Z","iopub.status.idle":"2022-08-08T14:03:00.387940Z","shell.execute_reply.started":"2022-08-08T14:03:00.312213Z","shell.execute_reply":"2022-08-08T14:03:00.386422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}