{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-17T15:00:14.450738Z","iopub.execute_input":"2023-09-17T15:00:14.451276Z","iopub.status.idle":"2023-09-17T15:00:14.460881Z","shell.execute_reply.started":"2023-09-17T15:00:14.451241Z","shell.execute_reply":"2023-09-17T15:00:14.459876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! unzip /kaggle/input/quora-insincere-questions-classification/embeddings.zip","metadata":{"execution":{"iopub.status.busy":"2023-09-17T15:00:17.449101Z","iopub.execute_input":"2023-09-17T15:00:17.449470Z","iopub.status.idle":"2023-09-17T15:02:12.091956Z","shell.execute_reply.started":"2023-09-17T15:00:17.449442Z","shell.execute_reply":"2023-09-17T15:02:12.090405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/train.csv')\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-17T15:02:19.239204Z","iopub.execute_input":"2023-09-17T15:02:19.239566Z","iopub.status.idle":"2023-09-17T15:02:23.495936Z","shell.execute_reply.started":"2023-09-17T15:02:19.239538Z","shell.execute_reply":"2023-09-17T15:02:23.494615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\n\nfrom tensorflow.keras.preprocessing.text import Tokenizer\n\ntokenizer = Tokenizer()\ntokenizer.fit_on_texts(data['question_text'].tolist())\ntext_seqs = tokenizer.texts_to_sequences(data['question_text'].tolist())\nlen(text_seqs)","metadata":{"execution":{"iopub.status.busy":"2023-09-17T15:02:23.498724Z","iopub.execute_input":"2023-09-17T15:02:23.499116Z","iopub.status.idle":"2023-09-17T15:03:24.552617Z","shell.execute_reply.started":"2023-09-17T15:02:23.499082Z","shell.execute_reply":"2023-09-17T15:03:24.551126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"emb_dict = dict()\nf = open('glove.840B.300d/glove.840B.300d.txt',encoding='unicode_escape')\nfor line in f:\n    tokens = line.split(' ')\n    emb_dict[tokens[0]] = np.array(tokens[1:],dtype=float)","metadata":{"execution":{"iopub.status.busy":"2023-09-17T15:04:15.329958Z","iopub.execute_input":"2023-09-17T15:04:15.330377Z","iopub.status.idle":"2023-09-17T15:06:27.464285Z","shell.execute_reply.started":"2023-09-17T15:04:15.330346Z","shell.execute_reply":"2023-09-17T15:06:27.463514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(emb_dict)","metadata":{"execution":{"iopub.status.busy":"2023-09-17T15:07:02.546959Z","iopub.execute_input":"2023-09-17T15:07:02.547365Z","iopub.status.idle":"2023-09-17T15:07:02.555800Z","shell.execute_reply.started":"2023-09-17T15:07:02.547335Z","shell.execute_reply":"2023-09-17T15:07:02.553818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class MyTransformer(keras.layers.Layer):\n    def __init__(self,embed_dim, num_heads, ff_dim):\n        super().__init__()\n        self.attn = keras.layers.MultiHeadAttention(num_heads=num_heads,key_dim=embed_dim)\n        self.ffn = keras.layers.Dense(embed_dim)\n        self.norm1 = keras.layers.LayerNormalization(epsilon=1e-6)\n        self.norm2 = keras.layers.LayerNormalization(epsilon=1e-6)\n        self.dropout1 = keras.layers.Dropout(0.1)\n        self.dropout2 = keras.layers.Dropout(0.1)\n    \n    def call(self,inputs):\n        attn_output = self.attn(inputs,inputs)\n        attn_output = self.dropout1(attn_output)\n        output1 = self.norm1(inputs+attn_output)\n        ffn_out = self.ffn(output1)\n        ffn_out = self.dropout2(ffn_out)\n        return self.norm2(ffn_out+output1)","metadata":{"execution":{"iopub.status.busy":"2023-09-17T15:07:08.375044Z","iopub.execute_input":"2023-09-17T15:07:08.375465Z","iopub.status.idle":"2023-09-17T15:07:08.382978Z","shell.execute_reply.started":"2023-09-17T15:07:08.375435Z","shell.execute_reply":"2023-09-17T15:07:08.381926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab = tokenizer.word_index\nnum_tokens = len(vocab)+2\nfinal_embed = np.zeros((num_tokens,300))\nfor i,word in enumerate(vocab):\n    if word in emb_dict.keys():\n        final_embed[i] = emb_dict[word]","metadata":{"execution":{"iopub.status.busy":"2023-09-17T15:07:08.384585Z","iopub.execute_input":"2023-09-17T15:07:08.384914Z","iopub.status.idle":"2023-09-17T15:07:08.907251Z","shell.execute_reply.started":"2023-09-17T15:07:08.384887Z","shell.execute_reply":"2023-09-17T15:07:08.905880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TokenPosEmbedding(keras.layers.Layer):\n    def __init__(self,embed_dim,max_len,embed_mat,num_tokens):\n        super().__init__()\n        self.token_emb = keras.layers.Embedding(num_tokens,embed_dim,\n                                                embeddings_initializer=keras.initializers.Constant(embed_mat),\n                                                trainable=False)\n        self.position_emb = keras.layers.Embedding(max_len,embed_dim)\n        \n    def call(self,x):\n        maxlen = tf.shape(x)[-1]\n        positions = tf.range(start=0, limit=maxlen, delta=1)\n        positions = self.position_emb(positions)\n        #pos_emb = self.position_emb([i for i in range(x.shape[1])])\n        tokenEmb = self.token_emb(x)\n        return positions+tokenEmb\n    ","metadata":{"execution":{"iopub.status.busy":"2023-09-17T15:07:08.908422Z","iopub.execute_input":"2023-09-17T15:07:08.908722Z","iopub.status.idle":"2023-09-17T15:07:08.915770Z","shell.execute_reply.started":"2023-09-17T15:07:08.908697Z","shell.execute_reply":"2023-09-17T15:07:08.914816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.sequence import pad_sequences\n\nmax_len = 100\ntext_seqs = pad_sequences(text_seqs,maxlen=max_len)","metadata":{"execution":{"iopub.status.busy":"2023-09-17T15:07:08.917954Z","iopub.execute_input":"2023-09-17T15:07:08.918239Z","iopub.status.idle":"2023-09-17T15:07:12.412606Z","shell.execute_reply.started":"2023-09-17T15:07:08.918214Z","shell.execute_reply":"2023-09-17T15:07:12.411104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs = keras.layers.Input(shape=(text_seqs.shape[1],))\nemb_layer = TokenPosEmbedding(300,100,final_embed,num_tokens)\nx = emb_layer(inputs)\ntransformer_layer = MyTransformer(300,2,32)\nx = transformer_layer(x)\nx = keras.layers.GlobalAvgPool1D()(x)\nx = keras.layers.Dropout(0.1)(x)\nx = keras.layers.Dense(32,activation=\"relu\")(x)\nx = keras.layers.Dropout(0.1)(x)\noutputs = outputs = keras.layers.Dense(2,activation='softmax')(x)\nmodel = keras.Model(inputs=inputs,outputs=outputs)\nmodel.compile(optimizer=\"adam\",loss=\"sparse_categorical_crossentropy\",metrics=[\"acc\"])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-09-17T15:07:12.413720Z","iopub.execute_input":"2023-09-17T15:07:12.414027Z","iopub.status.idle":"2023-09-17T15:07:13.253044Z","shell.execute_reply.started":"2023-09-17T15:07:12.414001Z","shell.execute_reply":"2023-09-17T15:07:13.251667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_X, test_X, train_y, test_y = train_test_split(text_seqs,data['target'])\nprint(len(train_X),len(test_X))\nprint(\"-----------------------\")\nprint(len(train_y),len(test_y))","metadata":{"execution":{"iopub.status.busy":"2023-09-17T15:07:13.254577Z","iopub.execute_input":"2023-09-17T15:07:13.254921Z","iopub.status.idle":"2023-09-17T15:07:14.257103Z","shell.execute_reply.started":"2023-09-17T15:07:13.254891Z","shell.execute_reply":"2023-09-17T15:07:14.255711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(train_X, train_y)","metadata":{"execution":{"iopub.status.busy":"2023-09-17T15:07:14.258850Z","iopub.execute_input":"2023-09-17T15:07:14.259198Z","iopub.status.idle":"2023-09-17T16:55:33.919110Z","shell.execute_reply.started":"2023-09-17T15:07:14.259172Z","shell.execute_reply":"2023-09-17T16:55:33.915008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\n\npreds = model.predict(test_X)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-09-17T17:14:44.743457Z","iopub.execute_input":"2023-09-17T17:14:44.743935Z","iopub.status.idle":"2023-09-17T17:28:53.759138Z","shell.execute_reply.started":"2023-09-17T17:14:44.743902Z","shell.execute_reply":"2023-09-17T17:28:53.758380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds","metadata":{"execution":{"iopub.status.busy":"2023-09-17T17:12:41.306337Z","iopub.execute_input":"2023-09-17T17:12:41.306852Z","iopub.status.idle":"2023-09-17T17:12:41.316899Z","shell.execute_reply.started":"2023-09-17T17:12:41.306815Z","shell.execute_reply":"2023-09-17T17:12:41.315871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_y","metadata":{"execution":{"iopub.status.busy":"2023-09-17T17:12:58.161056Z","iopub.execute_input":"2023-09-17T17:12:58.161476Z","iopub.status.idle":"2023-09-17T17:12:58.172033Z","shell.execute_reply.started":"2023-09-17T17:12:58.161447Z","shell.execute_reply":"2023-09-17T17:12:58.170687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_preds = [np.argmax(p) for p in preds]\n","metadata":{"execution":{"iopub.status.busy":"2023-09-17T17:14:21.247426Z","iopub.execute_input":"2023-09-17T17:14:21.247816Z","iopub.status.idle":"2023-09-17T17:14:22.001218Z","shell.execute_reply.started":"2023-09-17T17:14:21.247790Z","shell.execute_reply":"2023-09-17T17:14:22.000211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(roc_auc_score(test_y,y_preds))","metadata":{"execution":{"iopub.status.busy":"2023-09-17T17:14:37.924781Z","iopub.execute_input":"2023-09-17T17:14:37.925238Z","iopub.status.idle":"2023-09-17T17:14:38.005864Z","shell.execute_reply.started":"2023-09-17T17:14:37.925207Z","shell.execute_reply":"2023-09-17T17:14:38.003897Z"},"trusted":true},"execution_count":null,"outputs":[]}]}