{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import re\nfrom tqdm import tqdm\nfrom text_unidecode import unidecode\nimport string\n\nimport codecs\nfrom typing import Dict, List, Tuple\n\ndef replace_encoding_with_utf8(error: UnicodeError) -> Tuple[bytes, int]:\n    return error.object[error.start : error.end].encode(\"utf-8\"), error.end\n\n\ndef replace_decoding_with_cp1252(error: UnicodeError) -> Tuple[str, int]:\n    return error.object[error.start : error.end].decode(\"cp1252\"), error.end\n\ncodecs.register_error(\"replace_encoding_with_utf8\", replace_encoding_with_utf8)\ncodecs.register_error(\"replace_decoding_with_cp1252\", replace_decoding_with_cp1252)\n\ndef resolve_encodings_and_normalize(text: str) -> str:\n    text = (\n        text.encode(\"raw_unicode_escape\")\n        .decode(\"utf-8\", errors=\"replace_decoding_with_cp1252\")\n        .encode(\"cp1252\", errors=\"replace_encoding_with_utf8\")\n        .decode(\"utf-8\", errors=\"replace_decoding_with_cp1252\")\n    )\n    text = unidecode(text)\n    return text\n\nprint(resolve_encodings_and_normalize(\"asdasd adsasd\"))\nmisspell_mapping = {\n    'studentdesigned': 'student designed',\n    'teacherdesigned': 'teacher designed',\n    'genericname': 'generic name',\n    'winnertakeall': 'winner take all',\n    'studentname': 'student name',\n    'driveless': 'driverless',\n    'teachername': 'teacher name',\n    'propername': 'proper name',\n    'bestlaid': 'best laid',\n    'genericschool': 'generic school',\n    'schoolname': 'school name',\n    'winnertakesall': 'winner take all',\n    'elctoral': 'electoral',\n    'eletoral': 'electoral',\n    'genericcity': 'generic city',\n    'elctors': 'electoral',\n    'venuse': 'venue',\n    'blimplike': 'blimp like',\n    'selfdriving': 'self driving',\n    'electorals': 'electoral',\n    'nearrecord': 'near record',\n    'egyptianstyle': 'egyptian style',\n    'oddnumbered': 'odd numbered',\n    'carintensive': 'car intensive',\n    'elecoral': 'electoral',\n    'oction': 'auction',\n    'electroal': 'electoral',\n    'evennumbered': 'even numbered',\n    'mesalandforms': 'mesa landforms',\n    'electoralvote': 'electoral vote',\n    'relativename': 'relative name',\n    '22euro': 'twenty two euro',\n    'ellectoral': 'electoral',\n    'thirtyplus': 'thirty plus',\n    'collegewon': 'college won',\n    'hisher': 'higher',\n    'teacherbased': 'teacher based',\n    'computeranimated': 'computer animated',\n    'canadidate': 'candidate',\n    'studentbased': 'student based',\n    'gorethanks': 'gore thanks',\n    'clouddraped': 'cloud draped',\n    'edgarsnyder': 'edgar snyder',\n    'emotionrecognition': 'emotion recognition',\n    'landfrom': 'land form',\n    'fivedays': 'five days',\n    'electoal': 'electoral',\n    'lanform': 'land form',\n    'electral': 'electoral',\n    'presidentbut': 'president but',\n    'teacherassigned': 'teacher assigned',\n    'beacuas': 'because',\n    'positionestimating': 'position estimating',\n    'selfeducation': 'self education',\n    'diverless': 'driverless',\n    'computerdriven': 'computer driven',\n    'outofcontrol': 'out of control',\n    'faultthe': 'fault the',\n    'unfairoutdated': 'unfair outdated',\n    'aviods': 'avoid',\n    'momdad': 'mom dad',\n    'statesbig': 'states big',\n    'presidentswing': 'president swing',\n    'inconclusion': 'in conclusion',\n    'handsonlearning': 'hands on learning',\n    'electroral': 'electoral',\n    'carowner': 'car owner',\n    'elecotral': 'electoral',\n    'studentassigned': 'student assigned',\n    'collegefive': 'college five',\n    'presidant': 'president',\n    'unfairoutdatedand': 'unfair outdated and',\n    'nixonjimmy': 'nixon jimmy',\n    'canadates': 'candidate',\n    'tabletennis': 'table tennis',\n    'himher': 'him her',\n    'studentsummerpacketdesigners': 'student summer packet designers',\n    'studentdesign': 'student designed',\n    'limting': 'limiting',\n    'electrol': 'electoral',\n    'campaignto': 'campaign to',\n    'presendent': 'president',\n    'thezebra': 'the zebra',\n    'landformation': 'land formation',\n    'eyetoeye': 'eye to eye',\n    'selfreliance': 'self reliance',\n    'studentdriven': 'student driven',\n    'winnertake': 'winner take',\n    'alliens': 'aliens',\n    '2000but': '2000 but',\n    'electionto': 'election to',\n    'candidatesas': 'candidates as',\n    'electers': 'electoral',\n    'winnertakes': 'winner takes',\n    'isfeet': 'is feet',\n    'incar': 'incur',\n    'covid19': 'something',\n    'aflcio': '',\n    'outdatedand': 'outdated and',\n    'httpswww': '',\n    '51998': '',\n    'iswing': '',\n    'ascertainments': '',\n    'athome': '',\n    'risorius': '',\n    'votes538': '',\n    '41971': '',\n    'palpabraeus': '',\n    'figurelandform': 'figure landform',\n    'possibleit': 'possible it',\n    'takeall': 'take all',\n    'inschool': 'in school',\n    'fouces': 'focus',\n    'presidentand': 'president and',\n    'elecotrs': 'electoral',\n    'formationwhich': 'formation which',\n    'electorswho': 'electoral who',\n    'presidnt': 'president',\n    'eletors': 'electoral',\n    'sinceraly': 'sincerely',\n    'emotionshappiness': 'emotions happiness',\n    'carterbob': 'carter bob',\n    'donÃ£Ã¢t': 'do not',\n    'eyesnose': 'eyes nose',\n    'smartroad': 'smart road',\n    'systemvoters': 'system voters',\n    'emtions': 'emotions',\n    'statedemocrats': 'state democrats',\n    'lowcar': 'low car',\n    'elcetoral': 'electoral',\n    'expressivefor': 'expressive for',\n    'animails': 'animals',\n    'oppertonuty': 'opportunity',\n    'tempetures': 'temperature',\n    'recevies': 'receives',\n    'twoseat': 'two seat',\n    'consistution': 'constitution',\n    'horsesyoung': 'horses young',\n    'semidriverless': 'semi driverless',\n    'presisdent': 'president',\n    'exspression': 'expression',\n    'valcanoes': 'volcano',\n    'actiry': '',\n    'lifejust': 'life just',\n    'selfreliant': 'self reliant',\n    'comcaraccidentcauseofaccidentcellphonecellphonestatistics': 'car accident cause of accident cellphone statistics',\n    'vaubangermany': 'germany',\n    'fourtyfour': 'fourty four',\n    'atomspheric': 'atmospheric',\n    'mid1990': '',\n    'activitis': 'activities',\n    'paragrpah': 'paragraph',\n    'electora': 'electoral',\n    'elcetion': 'election',\n    'stressfree': 'stress free',\n    'seegoing': 'see going',\n    'coferencing': 'conferencing',\n    'ctrdot': '',\n    'segoing': '',\n    'teacherdesign': 'teacher design',\n    'kidsteens': 'kids teens',\n    'elcetors': 'electoral',\n    'poulltion': 'pollution',\n    'surportive': 'supportive',\n    'presisent': 'president',\n    'technollogy': 'technology',\n    'precidency': 'president',\n    'voteswhile': 'votes while',\n    'headformed': 'head formed',\n    'swingstates': 'swing states',\n    'candates': 'candidate',\n    'locationname': 'location name',\n    'venuss': 'venues',\n    'astronmers': 'astronomers',\n    'democtratic': 'democratic',\n    'canadent': 'candidate',\n    'cyndonia': '',\n    'computure': 'computer',\n    'nasas': 'nasa',\n    'onehalf': 'one half',\n    'preident': 'president',\n    'ressons': 'reasons',\n    'presidentvice': 'president vice',\n    'nonswing': 'non swing',\n    'thirtyeight': 'thirty eight',\n    'processnot': 'process not',\n    'facetoface': 'face to face',\n    'teendriversource': 'teen driver source',\n    'sadnessand': 'sadness and',\n    'abloish': 'abolish',\n    'driveing': 'driving',\n    'navagating': 'navigating',\n    'electorsthe': 'electoral',\n    'vothing': 'voting',\n    'callage': 'college',\n    'senseit': 'sense it',\n    'mercedesbenz': 'mercedes benz',\n    'electorall': 'electoral'\n}\ndef decontraction(phrase):\n    phrase = re.sub(r\"won\\'t\", \"will not\", phrase)\n    phrase = re.sub(r\"can\\'t\", \"can not\", phrase)\n    phrase = re.sub(r\"n\\'t\", \" not\", phrase)\n    phrase = re.sub(r\"\\'re\", \" are\", phrase)\n    phrase = re.sub(r\"\\'s\", \" is\", phrase)\n    phrase = re.sub(r\"\\'d\", \" would\", phrase)\n    phrase = re.sub(r\"\\'ll\", \" will\", phrase)\n    phrase = re.sub(r\"\\'t\", \" not\", phrase)\n    phrase = re.sub(r\"\\'ve\", \" have\", phrase)\n    phrase = re.sub(r\"\\'m\", \" am\", phrase)\n    phrase = re.sub(r\"he's\", \"he is\", phrase)\n    phrase = re.sub(r\"there's\", \"there is\", phrase)\n    phrase = re.sub(r\"We're\", \"We are\", phrase)\n    phrase = re.sub(r\"That's\", \"That is\", phrase)\n    phrase = re.sub(r\"won't\", \"will not\", phrase)\n    phrase = re.sub(r\"they're\", \"they are\", phrase)\n    phrase = re.sub(r\"Can't\", \"Cannot\", phrase)\n    phrase = re.sub(r\"wasn't\", \"was not\", phrase)\n    phrase = re.sub(r\"don\\x89Ûªt\", \"do not\", phrase)\n    phrase = re.sub(r\"donãât\", \"do not\", phrase)\n    phrase = re.sub(r\"aren't\", \"are not\", phrase)\n    phrase = re.sub(r\"isn't\", \"is not\", phrase)\n    phrase = re.sub(r\"What's\", \"What is\", phrase)\n    phrase = re.sub(r\"haven't\", \"have not\", phrase)\n    phrase = re.sub(r\"hasn't\", \"has not\", phrase)\n    phrase = re.sub(r\"There's\", \"There is\", phrase)\n    phrase = re.sub(r\"He's\", \"He is\", phrase)\n    phrase = re.sub(r\"It's\", \"It is\", phrase)\n    phrase = re.sub(r\"You're\", \"You are\", phrase)\n    phrase = re.sub(r\"I'M\", \"I am\", phrase)\n    phrase = re.sub(r\"shouldn't\", \"should not\", phrase)\n    phrase = re.sub(r\"wouldn't\", \"would not\", phrase)\n    phrase = re.sub(r\"i'm\", \"I am\", phrase)\n    phrase = re.sub(r\"I\\x89Ûªm\", \"I am\", phrase)\n    phrase = re.sub(r\"I'm\", \"I am\", phrase)\n    phrase = re.sub(r\"Isn't\", \"is not\", phrase)\n    phrase = re.sub(r\"Here's\", \"Here is\", phrase)\n    phrase = re.sub(r\"you've\", \"you have\", phrase)\n    phrase = re.sub(r\"you\\x89Ûªve\", \"you have\", phrase)\n    phrase = re.sub(r\"we're\", \"we are\", phrase)\n    phrase = re.sub(r\"what's\", \"what is\", phrase)\n    phrase = re.sub(r\"couldn't\", \"could not\", phrase)\n    phrase = re.sub(r\"we've\", \"we have\", phrase)\n    phrase = re.sub(r\"it\\x89Ûªs\", \"it is\", phrase)\n    phrase = re.sub(r\"doesn\\x89Ûªt\", \"does not\", phrase)\n    phrase = re.sub(r\"It\\x89Ûªs\", \"It is\", phrase)\n    phrase = re.sub(r\"Here\\x89Ûªs\", \"Here is\", phrase)\n    phrase = re.sub(r\"who's\", \"who is\", phrase)\n    phrase = re.sub(r\"I\\x89Ûªve\", \"I have\", phrase)\n    phrase = re.sub(r\"y'all\", \"you all\", phrase)\n    phrase = re.sub(r\"can\\x89Ûªt\", \"cannot\", phrase)\n    phrase = re.sub(r\"would've\", \"would have\", phrase)\n    phrase = re.sub(r\"it'll\", \"it will\", phrase)\n    phrase = re.sub(r\"we'll\", \"we will\", phrase)\n    phrase = re.sub(r\"wouldn\\x89Ûªt\", \"would not\", phrase)\n    phrase = re.sub(r\"We've\", \"We have\", phrase)\n    phrase = re.sub(r\"he'll\", \"he will\", phrase)\n    phrase = re.sub(r\"Y'all\", \"You all\", phrase)\n    phrase = re.sub(r\"Weren't\", \"Were not\", phrase)\n    phrase = re.sub(r\"Didn't\", \"Did not\", phrase)\n    phrase = re.sub(r\"they'll\", \"they will\", phrase)\n    phrase = re.sub(r\"they'd\", \"they would\", phrase)\n    phrase = re.sub(r\"DON'T\", \"DO NOT\", phrase)\n    phrase = re.sub(r\"That\\x89Ûªs\", \"That is\", phrase)\n    phrase = re.sub(r\"they've\", \"they have\", phrase)\n    phrase = re.sub(r\"i'd\", \"I would\", phrase)\n    phrase = re.sub(r\"should've\", \"should have\", phrase)\n    phrase = re.sub(r\"You\\x89Ûªre\", \"You are\", phrase)\n    phrase = re.sub(r\"where's\", \"where is\", phrase)\n    phrase = re.sub(r\"Don\\x89Ûªt\", \"Do not\", phrase)\n    phrase = re.sub(r\"we'd\", \"we would\", phrase)\n    phrase = re.sub(r\"i'll\", \"I will\", phrase)\n    phrase = re.sub(r\"weren't\", \"were not\", phrase)\n    phrase = re.sub(r\"They're\", \"They are\", phrase)\n    phrase = re.sub(r\"Can\\x89Ûªt\", \"Cannot\", phrase)\n    phrase = re.sub(r\"you\\x89Ûªll\", \"you will\", phrase)\n    phrase = re.sub(r\"I\\x89Ûªd\", \"I would\", phrase)\n    phrase = re.sub(r\"let's\", \"let us\", phrase)\n    phrase = re.sub(r\"it's\", \"it is\", phrase)\n    phrase = re.sub(r\"can't\", \"cannot\", phrase)\n    phrase = re.sub(r\"don't\", \"do not\", phrase)\n    phrase = re.sub(r\"you're\", \"you are\", phrase)\n    phrase = re.sub(r\"i've\", \"I have\", phrase)\n    phrase = re.sub(r\"that's\", \"that is\", phrase)\n    phrase = re.sub(r\"i'll\", \"I will\", phrase)\n    phrase = re.sub(r\"doesn't\", \"does not\",phrase)\n    phrase = re.sub(r\"i'd\", \"I would\", phrase)\n    phrase = re.sub(r\"didn't\", \"did not\", phrase)\n    phrase = re.sub(r\"ain't\", \"am not\", phrase)\n    phrase = re.sub(r\"you'll\", \"you will\", phrase)\n    phrase = re.sub(r\"I've\", \"I have\", phrase)\n    phrase = re.sub(r\"Don't\", \"do not\", phrase)\n    phrase = re.sub(r\"I'll\", \"I will\", phrase)\n    phrase = re.sub(r\"I'd\", \"I would\", phrase)\n    phrase = re.sub(r\"Let's\", \"Let us\", phrase)\n    phrase = re.sub(r\"you'd\", \"You would\", phrase)\n    phrase = re.sub(r\"It's\", \"It is\", phrase)\n    phrase = re.sub(r\"Ain't\", \"am not\", phrase)\n    phrase = re.sub(r\"Haven't\", \"Have not\", phrase)\n    phrase = re.sub(r\"Could've\", \"Could have\", phrase)\n    phrase = re.sub(r\"youve\", \"you have\", phrase)  \n    phrase = re.sub(r\"donå«t\", \"do not\", phrase)\n    return phrase\n\ndef clean_number(text):\n    text = re.sub(r'(\\d+)([a-zA-Z])', '\\g<1> \\g<2>', text)\n    text = re.sub(r'(\\d+) (th|st|nd|rd) ', '\\g<1>\\g<2> ', text)\n    text = re.sub(r'(\\d+),(\\d+)', '\\g<1>\\g<2>', text)\n    return text\n\ndef clean_misspell(text):\n    for bad_word in misspell_mapping:\n        if bad_word in text:\n            text = text.replace(bad_word, misspell_mapping[bad_word])\n    return text\n\ndef remove_punctuations(text):\n    for punctuation in list(string.punctuation):\n        text = text.replace(punctuation, '')\n    return text\n\ndef max_repeated_word_count(text):\n    words = [word for word in text.split() if word not in stopwords]\n\n    word_counts = Counter(words)\n    try:\n        return word_counts.most_common(1)[0][1]\n    \n    except IndexError:\n        return 0\n        \n    return max_count\n\ndef clean_text(text):\n    text = decontraction(text)\n    text = text.lower()\n    text = re.sub(r'[^\\w\\s]','',text, re.UNICODE)\n    text = remove_punctuations(text)\n    text = clean_number(text)\n    text = clean_misspell(text)\n    return text\n\nprint(clean_text(\"asdad\"))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T09:26:04.295579Z","iopub.execute_input":"2022-07-25T09:26:04.296601Z","iopub.status.idle":"2022-07-25T09:26:04.377456Z","shell.execute_reply.started":"2022-07-25T09:26:04.296551Z","shell.execute_reply":"2022-07-25T09:26:04.376432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport keras\nimport tensorflow as tf\nfrom tensorflow.keras.layers import TextVectorization\nfrom tensorflow.keras import layers\nimport re\nimport string\nfrom tensorflow.keras.preprocessing.text import one_hot\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow import keras\n\nfrom tqdm.notebook import tqdm\ntqdm.pandas()\nfrom stop_words import get_stop_words\nfrom collections import Counter\nfrom sklearn.preprocessing import MinMaxScaler\n\nscaler = MinMaxScaler()\n\nstopwords = list(get_stop_words('en'))\n\n\ndata = pd.read_csv(\"/kaggle/input/feedback-prize-effectiveness/train.csv\")\n#data = pd.concat([data,data.loc[data['discourse_effectiveness'] == \"Ineffective\"],data.loc[data['discourse_effectiveness'] == \"Ineffective\"],data.loc[data['discourse_effectiveness'] == \"Effective\"]], axis=0)\n#data= pd.concat([data.loc[data['discourse_effectiveness'] == \"Adequate\"].head(6000),data.loc[data['discourse_effectiveness'] == \"Ineffective\"].head(6000),data.loc[data['discourse_effectiveness'] == \"Effective\"].head(6000)])\n\"\"\"\ndata.discourse_effectiveness = data.loc[data['discourse_effectiveness'] == \"Adequate\"].head(6000)\ndata.discourse_effectiveness = data.loc[data['discourse_effectiveness'] == \"Ineffective\"].head(6000)\ndata.discourse_effectiveness = data.loc[data['discourse_effectiveness'] == \"Effective\"].head(6000)\n\"\"\"\n\ndata['discourse_text'] = data['discourse_text'].apply(lambda x : resolve_encodings_and_normalize(x))\n\ndata['contains_source'] = data['discourse_text'].apply(lambda x: 'source' in x.lower().split())\ndata['contains_I'] = data['discourse_text'].apply(lambda x: 'i' in x.lower().split())\ndata['max_repeated_word_count'] = data['discourse_text'].apply(max_repeated_word_count)\ndisplay(data['max_repeated_word_count'][15:35])\n\ndata['max_repeated_word_count'] = scaler.fit_transform(data['max_repeated_word_count'].values.reshape(-1, 1)).reshape(-1)\ndisplay(data['max_repeated_word_count'][0])\nprint(\"{:.8f}\".format(data['max_repeated_word_count'][0]))\ndisplay(data.discourse_effectiveness.value_counts())\n\nprint(data['max_repeated_word_count'].values.reshape (-1,1).reshape(-1))\nmax_len = data.discourse_text.str.split().str.len().max()\n\neffectiveness = np.expand_dims(data.discourse_effectiveness.values, axis=1)\n\n\nprint(data.discourse_text.str.split())\nprint(max_len)\n\nessays = []\n\n\nfor i in data.essay_id.unique():\n    f=open(\"/kaggle/input/feedback-prize-effectiveness/train/{}.txt\".format(i),'r')\n    essays.append(f.read())\n    f.close()\n    \nessays_str = ''.join(essays)\nuniq_vocab_len = len(set(essays_str.split()))\nprint(type(essays_str))\n\ndef custom_standardization(input_data):\n  lowercase = tf.strings.lower(input_data)\n  stripped_html = tf.strings.regex_replace(lowercase, '<br />', ' ')\n  return tf.strings.regex_replace(stripped_html,'[%s]' % re.escape(string.punctuation), '')\n\n\nvocab_size = 10000\nsequence_length = 836\n\n\nvectorize_layer = TextVectorization(\n    standardize=\"lower_and_strip_punctuation\",\n    max_tokens=vocab_size,\n    output_mode='int',\n    output_sequence_length=sequence_length)\n\ndata['discourse_text'] = data['discourse_text'].progress_apply(clean_text)\nvectorize_layer.adapt(list(data.discourse_text))\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-25T09:26:04.379678Z","iopub.execute_input":"2022-07-25T09:26:04.380311Z","iopub.status.idle":"2022-07-25T09:26:50.232737Z","shell.execute_reply.started":"2022-07-25T09:26:04.380268Z","shell.execute_reply":"2022-07-25T09:26:50.231707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_position_encoding(seq_len, d, n=10000):\n    P = np.zeros((seq_len, d))\n    for k in range(seq_len):\n        for i in np.arange(int(d/2)):\n            denominator = np.power(n, 2*i/d)\n            P[k, 2*i] = np.sin(k/denominator)\n            P[k, 2*i+1] = np.cos(k/denominator)\n    return P\n\ndef network():\n    \n    inp = layers.Input(shape=(1,), dtype=tf.string)\n    inp2 = layers.Input(shape=(10))\n    \n    x = vectorize_layer(inp)\n    embedded =  tf.keras.layers.Embedding(\n        input_dim=len(vectorize_layer.get_vocabulary()),\n        output_dim=100,\n        mask_zero=True,\n        weights=[get_position_encoding(len(vectorize_layer.get_vocabulary()), 100)],\n        trainable=False)(x);\n    \n    \n    \n    \n    positions = tf.range(start=0, limit=836, delta=1)\n    positions = tf.keras.layers.Embedding(input_dim=836, output_dim=100, weights=[get_position_encoding(836, 100)],\n            trainable=False)(positions)\n    \n    embedded= embedded+positions\n    \n    x = layers.RepeatVector(836)(inp2)\n\n    print(embedded.shape)\n    \n    #embedded = tf.keras.layers.Concatenate()([embedded, x])\n\n    att1=layers.MultiHeadAttention(key_dim=100 , num_heads=8, dropout=0.25)(embedded, x)\n \n    out1=layers.LayerNormalization(epsilon=1e-6)(embedded+att1)\n\n    #x = layers.Conv1D(filters=74, padding='same' , kernel_size=32, activation=\"elu\")(att1)\n    x=keras.layers.Dense(128,activation=\"relu\")(out1)\n    x=keras.layers.Dense(100)(x)\n\n    x = layers.Dropout(0.2)(x)\n\n        \n    out2 = layers.LayerNormalization(epsilon=1e-6)(x+out1)\n\n    \n    att2=layers.MultiHeadAttention(key_dim=100, num_heads=8, dropout=0.25)(x, x)\n\n    out3=layers.LayerNormalization(epsilon=1e-6)(out2+att2)\n\n    \n    #x = layers.Conv1D(filters=74, padding='same', kernel_size=8, activation=\"elu\")(x)\n    x=keras.layers.Dense(128,activation=\"relu\")(out3)\n    x=keras.layers.Dense(100)(x)\n\n    \n    x = layers.Dropout(0.2)(x)\n    \n        \n    out4 = layers.LayerNormalization(epsilon=1e-6)(x+out3)\n\n    \n    att3=layers.MultiHeadAttention(key_dim=100, num_heads=8, dropout=0.25)(x, x)\n\n    out5=layers.LayerNormalization(epsilon=1e-6)(out4+att3)\n\n    #x = layers.Conv1D(filters=74, padding='same', kernel_size=8, activation=\"elu\")(x)\n    \n    x=keras.layers.Dense(128,activation=\"relu\")(out5)\n    x=keras.layers.Dense(100)(x)\n\n    x = layers.Dropout(0.2)(x)\n    \n    \n    x = layers.LayerNormalization(epsilon=1e-6)(x+out5)\n\n    #x=layers.MultiHeadAttention(key_dim=2, num_heads=2, dropout=0.25)(x, x)\n\n    x = layers.GlobalAveragePooling1D()(x)\n    #x=tf.keras.layers.Flatten()(x)\n\n    x = tf.keras.layers.BatchNormalization()(x)\n\n    \n    x=keras.layers.Dense(32,activation=\"relu\")(x)\n\n    x = layers.Dropout(0.2)(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n\n\n    \n    out=keras.layers.Dense(3,activation=\"softmax\")(x)\n    \n    model = keras.Model(inputs=[inp,inp2], outputs=out)\n    return model\n\nmodel=network()\nmodel.compile(optimizer=keras.optimizers.Adam(learning_rate=0.0003),\n              loss=tf.keras.losses.CategoricalCrossentropy(from_logits=False),\n              metrics=['accuracy'])\n\nprint(model.summary())\ninp = np.array(list(data.discourse_text))\ninp2 = pd.get_dummies(data.discourse_type.values)\n\n\n\ninp2 = pd.concat([inp2,data['contains_source'].astype(int),data['contains_I'].astype(int),data['max_repeated_word_count']], ignore_index=False ,axis=1).values\ndisplay(inp2)\nout = pd.get_dummies(data.discourse_effectiveness.values).values\nprint(out[10])\nprint(data.discourse_effectiveness.values[10])\nX1_train, X1_test, X2_train, X2_test, y_train, y_test = train_test_split(inp,inp2, out, test_size=0.1, random_state=42)\n\n\n\nprint(inp2.shape)\nprint(y_test.shape)\n\n\nprint(X2_train[0])\n\n\n\ndef decay_schedule(epoch, lr):\n    # decay by 0.1 every 5 epochs; use `% 1` to decay after each epoch\n    print(lr)\n    \n    if (epoch % 2 == 0) and (epoch != 0):\n        lr = lr * 0.9\n    return lr\n\n\n\ncallbacks = [\n    keras.callbacks.EarlyStopping(\n    monitor='val_loss',\n    patience=20,\n    restore_best_weights=True)\n]\n\n\nmodel.fit(x=[X1_train,X2_train],y=y_train,epochs=2,validation_data=([X1_test,X2_test],y_test),batch_size=32,validation_split=0.3,callbacks=[callbacks])\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-25T09:30:33.250053Z","iopub.execute_input":"2022-07-25T09:30:33.250492Z","iopub.status.idle":"2022-07-25T09:30:57.377803Z","shell.execute_reply.started":"2022-07-25T09:30:33.250459Z","shell.execute_reply":"2022-07-25T09:30:57.375580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_test = pd.read_csv(\"/kaggle/input/feedback-prize-effectiveness/test.csv\")\n\ndata_test['discourse_text'] = data['discourse_text'].apply(lambda x : resolve_encodings_and_normalize(x))\n\ndata_test['contains_source'] = data['discourse_text'].apply(lambda x: 'source' in x.lower().split())\ndata_test['contains_I'] = data['discourse_text'].apply(lambda x: 'i' in x.lower().split())\ndata_test['max_repeated_word_count'] = data['discourse_text'].apply(max_repeated_word_count)\ndata_test['max_repeated_word_count'] = scaler.transform(data_test['max_repeated_word_count'].values.reshape(-1, 1)).reshape(-1)\n\ndata_test['discourse_text'] = data_test['discourse_text'].progress_apply(clean_text)\n\ninp_test = np.array(list(data_test.discourse_text))\nprint(data_test.shape[0])\ninp2_test = pd.get_dummies(np.append([\"Lead\",\"Position\",\"Claim\",\"Evidence\",\"Counterclaim\",\"Rebuttal\",\"Concluding Statement\"],data_test.discourse_type.values))[-data_test.shape[0]:]\ninp2_test = inp2_test.reset_index(drop=True)\n\ninp2_test = pd.concat([inp2_test,data_test['contains_source'].astype(int),data_test['contains_I'].astype(int),data_test['max_repeated_word_count']], ignore_index=False ,axis=1).values\ndisplay(inp2_test)\n\nprint(inp2_test.shape)\npredict = model.predict([inp_test,inp2_test] , verbose=1)\nsub = pd.read_csv(\"/kaggle/input/feedback-prize-effectiveness/sample_submission.csv\")\n\nsub['Ineffective'] = predict[:,2]\nsub['Adequate'] = predict[:,0]\nsub['Effective'] = predict[:,1]\n\ndisplay(sub)\nsub.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T09:29:41.846302Z","iopub.execute_input":"2022-07-25T09:29:41.847129Z","iopub.status.idle":"2022-07-25T09:29:47.209588Z","shell.execute_reply.started":"2022-07-25T09:29:41.847089Z","shell.execute_reply":"2022-07-25T09:29:47.208510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}