{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-09-12T09:09:54.589737Z","iopub.execute_input":"2021-09-12T09:09:54.590148Z","iopub.status.idle":"2021-09-12T09:09:54.683874Z","shell.execute_reply.started":"2021-09-12T09:09:54.59003Z","shell.execute_reply":"2021-09-12T09:09:54.682434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!unzip /kaggle/input/quora-insincere-questions-classification/embeddings.zip","metadata":{"execution":{"iopub.status.busy":"2021-09-12T09:45:59.051588Z","iopub.execute_input":"2021-09-12T09:45:59.051911Z","iopub.status.idle":"2021-09-12T09:49:39.572421Z","shell.execute_reply.started":"2021-09-12T09:45:59.051865Z","shell.execute_reply":"2021-09-12T09:49:39.571267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport time\nimport math\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nfrom sklearn import metrics\nfrom sklearn.model_selection import train_test_split\n\n#below are TF Imports\nfrom tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom tensorflow.keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, Conv1D\nfrom tensorflow.compat.v1.keras.layers import CuDNNGRU # compatible with tf2.0\nfrom tensorflow.keras.layers import Bidirectional, GlobalMaxPool1D\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras import initializers, regularizers, constraints, optimizers, layers","metadata":{"execution":{"iopub.status.busy":"2021-09-12T09:17:20.150431Z","iopub.execute_input":"2021-09-12T09:17:20.150717Z","iopub.status.idle":"2021-09-12T09:17:20.238175Z","shell.execute_reply.started":"2021-09-12T09:17:20.150689Z","shell.execute_reply":"2021-09-12T09:17:20.237302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/test.csv\")\nprint(\"Train shape : \",train_df.shape)\nprint(\"Test shape : \",test_df.shape)","metadata":{"execution":{"iopub.status.busy":"2021-09-12T09:10:57.849031Z","iopub.execute_input":"2021-09-12T09:10:57.849595Z","iopub.status.idle":"2021-09-12T09:11:03.522725Z","shell.execute_reply.started":"2021-09-12T09:10:57.849564Z","shell.execute_reply":"2021-09-12T09:11:03.521731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## split to train and val\ntrain_df, val_df = train_test_split(train_df, test_size=0.1, random_state=2018)\n\n## some config values \nembed_size = 300 # how big is each word vector\nmax_features = 50000 # how many unique words to use (i.e num rows in embedding vector)\nmaxlen = 100 # max number of words in a question to use\n\n## fill up the missing values\ntrain_X = train_df[\"question_text\"].fillna(\"_na_\").values\nval_X = val_df[\"question_text\"].fillna(\"_na_\").values\ntest_X = test_df[\"question_text\"].fillna(\"_na_\").values\n\n## Tokenize the sentences\ntokenizer = Tokenizer(num_words=max_features)\ntokenizer.fit_on_texts(list(train_X))\ntrain_X = tokenizer.texts_to_sequences(train_X)\nval_X = tokenizer.texts_to_sequences(val_X)\ntest_X = tokenizer.texts_to_sequences(test_X)\n\n## Pad the sentences \ntrain_X = pad_sequences(train_X, maxlen=maxlen)\nval_X = pad_sequences(val_X, maxlen=maxlen)\ntest_X = pad_sequences(test_X, maxlen=maxlen)\n\n## Get the target values\ntrain_y = train_df['target'].values\nval_y = val_df['target'].values","metadata":{"execution":{"iopub.status.busy":"2021-09-12T09:11:25.325436Z","iopub.execute_input":"2021-09-12T09:11:25.325811Z","iopub.status.idle":"2021-09-12T09:12:38.83933Z","shell.execute_reply.started":"2021-09-12T09:11:25.32576Z","shell.execute_reply":"2021-09-12T09:12:38.838395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inp = Input(shape=(maxlen,))\nx = Embedding(max_features, embed_size)(inp)\nx = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\nx = GlobalMaxPool1D()(x)\nx = Dense(16, activation=\"relu\")(x)\nx = Dropout(0.1)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n\nprint(model.summary())","metadata":{"execution":{"iopub.status.busy":"2021-09-12T09:17:56.615603Z","iopub.execute_input":"2021-09-12T09:17:56.615878Z","iopub.status.idle":"2021-09-12T09:17:59.579712Z","shell.execute_reply.started":"2021-09-12T09:17:56.61585Z","shell.execute_reply":"2021-09-12T09:17:59.578666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Train the model \nmodel.fit(train_X, train_y, batch_size=1024, epochs=2, validation_data=(val_X, val_y))","metadata":{"execution":{"iopub.status.busy":"2021-09-12T09:18:24.954451Z","iopub.execute_input":"2021-09-12T09:18:24.954745Z","iopub.status.idle":"2021-09-12T09:42:25.017468Z","shell.execute_reply.started":"2021-09-12T09:18:24.954718Z","shell.execute_reply":"2021-09-12T09:42:25.016432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_no_emb_val_y = model.predict([val_X], batch_size=1024, verbose=1)\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_no_emb_val_y>thresh).astype(int))))","metadata":{"execution":{"iopub.status.busy":"2021-09-12T09:42:49.528489Z","iopub.execute_input":"2021-09-12T09:42:49.528851Z","iopub.status.idle":"2021-09-12T09:42:57.464058Z","shell.execute_reply.started":"2021-09-12T09:42:49.528805Z","shell.execute_reply":"2021-09-12T09:42:57.46302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_no_emb_test_y = model.predict([test_X], batch_size=1024, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2021-09-12T09:43:05.026891Z","iopub.execute_input":"2021-09-12T09:43:05.027539Z","iopub.status.idle":"2021-09-12T09:43:22.192486Z","shell.execute_reply.started":"2021-09-12T09:43:05.02749Z","shell.execute_reply":"2021-09-12T09:43:22.19153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del model, inp, x\nimport gc; gc.collect()\ntime.sleep(10)","metadata":{"execution":{"iopub.status.busy":"2021-09-12T09:43:35.999093Z","iopub.execute_input":"2021-09-12T09:43:35.999398Z","iopub.status.idle":"2021-09-12T09:43:46.25196Z","shell.execute_reply.started":"2021-09-12T09:43:35.99937Z","shell.execute_reply":"2021-09-12T09:43:46.250782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls ../input/embeddings/\n","metadata":{"execution":{"iopub.status.busy":"2021-09-12T09:44:12.534978Z","iopub.execute_input":"2021-09-12T09:44:12.535313Z","iopub.status.idle":"2021-09-12T09:44:13.377319Z","shell.execute_reply.started":"2021-09-12T09:44:12.535284Z","shell.execute_reply":"2021-09-12T09:44:13.376124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Glove Embeddings:","metadata":{"execution":{"iopub.status.busy":"2021-09-12T09:44:28.359797Z","iopub.execute_input":"2021-09-12T09:44:28.360104Z","iopub.status.idle":"2021-09-12T09:44:28.364856Z","shell.execute_reply.started":"2021-09-12T09:44:28.360072Z","shell.execute_reply":"2021-09-12T09:44:28.363578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EMBEDDING_FILE = './glove.840B.300d/glove.840B.300d.txt'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE))\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n        \ninp = Input(shape=(maxlen,))\nx = Embedding(max_features, embed_size, weights=[embedding_matrix])(inp)\nx = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\nx = GlobalMaxPool1D()(x)\nx = Dense(16, activation=\"relu\")(x)\nx = Dropout(0.1)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\nprint(model.summary())","metadata":{"execution":{"iopub.status.busy":"2021-09-12T09:49:39.577017Z","iopub.execute_input":"2021-09-12T09:49:39.577314Z","iopub.status.idle":"2021-09-12T09:54:41.253069Z","shell.execute_reply.started":"2021-09-12T09:49:39.577279Z","shell.execute_reply":"2021-09-12T09:54:41.252117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(train_X, train_y, batch_size=1024, epochs=2, validation_data=(val_X, val_y))","metadata":{"execution":{"iopub.status.busy":"2021-09-12T10:08:22.826404Z","iopub.execute_input":"2021-09-12T10:08:22.82669Z","iopub.status.idle":"2021-09-12T10:33:48.547713Z","shell.execute_reply.started":"2021-09-12T10:08:22.826661Z","shell.execute_reply":"2021-09-12T10:33:48.546661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_glove_val_y = model.predict([val_X], batch_size=1024, verbose=1)\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_glove_val_y>thresh).astype(int))))","metadata":{"execution":{"iopub.status.busy":"2021-09-12T10:37:25.684655Z","iopub.execute_input":"2021-09-12T10:37:25.685193Z","iopub.status.idle":"2021-09-12T10:37:33.8329Z","shell.execute_reply.started":"2021-09-12T10:37:25.68516Z","shell.execute_reply":"2021-09-12T10:37:33.83183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_glove_test_y = model.predict([test_X], batch_size=1024, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2021-09-12T10:37:37.629451Z","iopub.execute_input":"2021-09-12T10:37:37.630319Z","iopub.status.idle":"2021-09-12T10:37:55.066915Z","shell.execute_reply.started":"2021-09-12T10:37:37.630287Z","shell.execute_reply":"2021-09-12T10:37:55.065891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del word_index, embeddings_index, all_embs, embedding_matrix, model, inp, x\nimport gc; gc.collect()\ntime.sleep(10)","metadata":{"execution":{"iopub.status.busy":"2021-09-12T10:38:05.790074Z","iopub.execute_input":"2021-09-12T10:38:05.790983Z","iopub.status.idle":"2021-09-12T10:38:17.205649Z","shell.execute_reply.started":"2021-09-12T10:38:05.79095Z","shell.execute_reply":"2021-09-12T10:38:17.204517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EMBEDDING_FILE = './wiki-news-300d-1M/wiki-news-300d-1M.vec'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE) if len(o)>100)\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n        \ninp = Input(shape=(maxlen,))\nx = Embedding(max_features, embed_size, weights=[embedding_matrix])(inp)\nx = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\nx = GlobalMaxPool1D()(x)\nx = Dense(16, activation=\"relu\")(x)\nx = Dropout(0.1)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2021-09-12T10:40:32.66253Z","iopub.execute_input":"2021-09-12T10:40:32.662945Z","iopub.status.idle":"2021-09-12T10:42:40.291867Z","shell.execute_reply.started":"2021-09-12T10:40:32.662914Z","shell.execute_reply":"2021-09-12T10:42:40.290319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(train_X, train_y, batch_size=1024, epochs=2, validation_data=(val_X, val_y))","metadata":{"execution":{"iopub.status.busy":"2021-09-12T10:44:09.145421Z","iopub.execute_input":"2021-09-12T10:44:09.145772Z","iopub.status.idle":"2021-09-12T10:58:49.337051Z","shell.execute_reply.started":"2021-09-12T10:44:09.145741Z","shell.execute_reply":"2021-09-12T10:58:49.336035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_fasttext_val_y = model.predict([val_X], batch_size=1024, verbose=1)\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_fasttext_val_y>thresh).astype(int))))","metadata":{"execution":{"iopub.status.busy":"2021-09-12T10:58:53.895796Z","iopub.execute_input":"2021-09-12T10:58:53.896744Z","iopub.status.idle":"2021-09-12T10:59:02.227918Z","shell.execute_reply.started":"2021-09-12T10:58:53.896704Z","shell.execute_reply":"2021-09-12T10:59:02.226843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"References from https://www.kaggle.com/sudalairajkumar/a-look-at-different-embeddings","metadata":{}}]}