{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-17T07:51:11.842735Z","iopub.execute_input":"2023-09-17T07:51:11.843456Z","iopub.status.idle":"2023-09-17T07:51:11.853140Z","shell.execute_reply.started":"2023-09-17T07:51:11.843418Z","shell.execute_reply":"2023-09-17T07:51:11.852063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! unzip /kaggle/input/quora-insincere-questions-classification/embeddings.zip","metadata":{"execution":{"iopub.status.busy":"2023-09-17T07:51:11.855459Z","iopub.execute_input":"2023-09-17T07:51:11.855977Z","iopub.status.idle":"2023-09-17T07:51:29.561268Z","shell.execute_reply.started":"2023-09-17T07:51:11.855937Z","shell.execute_reply":"2023-09-17T07:51:29.560093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/train.csv')\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-17T07:51:29.564293Z","iopub.execute_input":"2023-09-17T07:51:29.565160Z","iopub.status.idle":"2023-09-17T07:51:34.531728Z","shell.execute_reply.started":"2023-09-17T07:51:29.565076Z","shell.execute_reply":"2023-09-17T07:51:34.530940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\n\nfrom tensorflow.keras.preprocessing.text import Tokenizer\n\ntokenizer = Tokenizer()\ntokenizer.fit_on_texts(data['question_text'].tolist())\ntext_seqs = tokenizer.texts_to_sequences(data['question_text'].tolist())\nlen(text_seqs)","metadata":{"execution":{"iopub.status.busy":"2023-09-17T07:51:34.533097Z","iopub.execute_input":"2023-09-17T07:51:34.534255Z","iopub.status.idle":"2023-09-17T07:52:42.882577Z","shell.execute_reply.started":"2023-09-17T07:51:34.534217Z","shell.execute_reply":"2023-09-17T07:52:42.881433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"param_dict = dict()\nf = open('paragram_300_sl999/paragram_300_sl999.txt',encoding='unicode_escape')\nfor line in f:\n    tokens = line.split(' ')\n    param_dict[tokens[0]] = np.array(tokens[1:],dtype=float)","metadata":{"execution":{"iopub.status.busy":"2023-09-17T07:52:42.885843Z","iopub.execute_input":"2023-09-17T07:52:42.886755Z","iopub.status.idle":"2023-09-17T07:55:42.913537Z","shell.execute_reply.started":"2023-09-17T07:52:42.886706Z","shell.execute_reply":"2023-09-17T07:55:42.912422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(param_dict)","metadata":{"execution":{"iopub.status.busy":"2023-09-17T07:55:42.914937Z","iopub.execute_input":"2023-09-17T07:55:42.915286Z","iopub.status.idle":"2023-09-17T07:55:42.922270Z","shell.execute_reply.started":"2023-09-17T07:55:42.915259Z","shell.execute_reply":"2023-09-17T07:55:42.921059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class MyTransformer(keras.layers.Layer):\n    def __init__(self,embed_dim, num_heads, ff_dim):\n        super().__init__()\n        self.attn = keras.layers.MultiHeadAttention(num_heads=num_heads,key_dim=embed_dim)\n        self.ffn = keras.layers.Dense(embed_dim)\n        self.norm1 = keras.layers.LayerNormalization(epsilon=1e-6)\n        self.norm2 = keras.layers.LayerNormalization(epsilon=1e-6)\n        self.dropout1 = keras.layers.Dropout(0.1)\n        self.dropout2 = keras.layers.Dropout(0.1)\n    \n    def call(self,inputs):\n        attn_output = self.attn(inputs,inputs)\n        attn_output = self.dropout1(attn_output)\n        output1 = self.norm1(inputs+attn_output)\n        ffn_out = self.ffn(output1)\n        ffn_out = self.dropout2(ffn_out)\n        return self.norm2(ffn_out+output1)","metadata":{"execution":{"iopub.status.busy":"2023-09-17T07:55:42.923477Z","iopub.execute_input":"2023-09-17T07:55:42.923792Z","iopub.status.idle":"2023-09-17T07:55:42.948964Z","shell.execute_reply.started":"2023-09-17T07:55:42.923763Z","shell.execute_reply":"2023-09-17T07:55:42.947855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab = tokenizer.word_index\nnum_tokens = len(vocab)+2\nfinal_embed = np.zeros((num_tokens,300))\nfor i,word in enumerate(vocab):\n    if word in param_dict.keys():\n        final_embed[i] = param_dict[word]","metadata":{"execution":{"iopub.status.busy":"2023-09-17T07:55:42.950673Z","iopub.execute_input":"2023-09-17T07:55:42.951041Z","iopub.status.idle":"2023-09-17T07:55:44.000458Z","shell.execute_reply.started":"2023-09-17T07:55:42.951012Z","shell.execute_reply":"2023-09-17T07:55:43.999254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TokenPosEmbedding(keras.layers.Layer):\n    def __init__(self,embed_dim,max_len,embed_mat,num_tokens):\n        super().__init__()\n        self.token_emb = keras.layers.Embedding(num_tokens,embed_dim,\n                                                embeddings_initializer=keras.initializers.Constant(embed_mat),\n                                                trainable=False)\n        self.position_emb = keras.layers.Embedding(max_len,embed_dim)\n        \n    def call(self,x):\n        maxlen = tf.shape(x)[-1]\n        positions = tf.range(start=0, limit=maxlen, delta=1)\n        positions = self.position_emb(positions)\n        #pos_emb = self.position_emb([i for i in range(x.shape[1])])\n        tokenEmb = self.token_emb(x)\n        return positions+tokenEmb\n    ","metadata":{"execution":{"iopub.status.busy":"2023-09-17T07:55:44.002162Z","iopub.execute_input":"2023-09-17T07:55:44.002518Z","iopub.status.idle":"2023-09-17T07:55:44.010982Z","shell.execute_reply.started":"2023-09-17T07:55:44.002487Z","shell.execute_reply":"2023-09-17T07:55:44.010168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.sequence import pad_sequences\n\nmax_len = 100\ntext_seqs = pad_sequences(text_seqs,maxlen=max_len)","metadata":{"execution":{"iopub.status.busy":"2023-09-17T07:55:44.012333Z","iopub.execute_input":"2023-09-17T07:55:44.012642Z","iopub.status.idle":"2023-09-17T07:55:51.606846Z","shell.execute_reply.started":"2023-09-17T07:55:44.012614Z","shell.execute_reply":"2023-09-17T07:55:51.605884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs = keras.layers.Input(shape=(text_seqs.shape[1],))\nemb_layer = TokenPosEmbedding(300,100,final_embed,num_tokens)\nx = emb_layer(inputs)\ntransformer_layer = MyTransformer(300,2,32)\nx = transformer_layer(x)\nx = keras.layers.GlobalAvgPool1D()(x)\nx = keras.layers.Dropout(0.1)(x)\nx = keras.layers.Dense(32,activation=\"relu\")(x)\nx = keras.layers.Dropout(0.1)(x)\noutputs = outputs = keras.layers.Dense(2,activation='softmax')(x)\nmodel = keras.Model(inputs=inputs,outputs=outputs)\nmodel.compile(optimizer=\"adam\",loss=\"sparse_categorical_crossentropy\",metrics=[\"acc\"])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-09-17T07:55:51.609785Z","iopub.execute_input":"2023-09-17T07:55:51.610093Z","iopub.status.idle":"2023-09-17T07:55:52.714344Z","shell.execute_reply.started":"2023-09-17T07:55:51.610065Z","shell.execute_reply":"2023-09-17T07:55:52.713198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_X, test_X, train_y, test_y = train_test_split(text_seqs,data['target'])\nprint(len(train_X),len(test_X))\nprint(\"-----------------------\")\nprint(len(train_y),len(test_y))","metadata":{"execution":{"iopub.status.busy":"2023-09-17T07:55:52.715961Z","iopub.execute_input":"2023-09-17T07:55:52.716867Z","iopub.status.idle":"2023-09-17T07:55:54.038722Z","shell.execute_reply.started":"2023-09-17T07:55:52.716834Z","shell.execute_reply":"2023-09-17T07:55:54.037492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(train_X, train_y)","metadata":{"execution":{"iopub.status.busy":"2023-09-17T07:55:54.039970Z","iopub.execute_input":"2023-09-17T07:55:54.040285Z"},"trusted":true},"execution_count":null,"outputs":[]}]}