{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-05-27T15:59:45.812898Z","iopub.execute_input":"2021-05-27T15:59:45.813210Z","iopub.status.idle":"2021-05-27T15:59:45.822712Z","shell.execute_reply.started":"2021-05-27T15:59:45.813181Z","shell.execute_reply":"2021-05-27T15:59:45.821765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission_filepath = \"/kaggle/input/quora-insincere-questions-classification/sample_submission.csv\"\nembeddings_zippath = \"/kaggle/input/quora-insincere-questions-classification/embeddings.zip\"\ntrain_csv_path = \"/kaggle/input/quora-insincere-questions-classification/train.csv\"\ntest_csv_path = \"/kaggle/input/quora-insincere-questions-classification/test.csv\"","metadata":{"execution":{"iopub.status.busy":"2021-05-27T15:59:46.029691Z","iopub.execute_input":"2021-05-27T15:59:46.030112Z","iopub.status.idle":"2021-05-27T15:59:46.035994Z","shell.execute_reply.started":"2021-05-27T15:59:46.030073Z","shell.execute_reply":"2021-05-27T15:59:46.034943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(train_csv_path)\ntest_df = pd.read_csv(test_csv_path)","metadata":{"execution":{"iopub.status.busy":"2021-05-27T15:59:46.414162Z","iopub.execute_input":"2021-05-27T15:59:46.414482Z","iopub.status.idle":"2021-05-27T15:59:50.814286Z","shell.execute_reply.started":"2021-05-27T15:59:46.414453Z","shell.execute_reply":"2021-05-27T15:59:50.813405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have four different types of embeddings.\n\n* GoogleNews-vectors-negative300 - https://code.google.com/archive/p/word2vec/\n* glove.840B.300d - https://nlp.stanford.edu/projects/glove/\n* paragram_300_sl999 - https://cogcomp.org/page/resource_view/106\n* wiki-news-300d-1M - https://fasttext.cc/docs/en/english-vectors.html\n\nA very good explanation for different types of embeddings are given in this kernel. Please refer the same for more details..\n\n# Glove Embeddings:\n\n    In this section, let us use the Glove embeddings and rebuild the GRU model.","metadata":{}},{"cell_type":"code","source":"# unzip file.zip -d destination_folder\n!unzip /kaggle/input/quora-insincere-questions-classification/embeddings.zip -d /kaggle/working/embeddings","metadata":{"execution":{"iopub.status.busy":"2021-05-27T15:59:50.815674Z","iopub.execute_input":"2021-05-27T15:59:50.816243Z","iopub.status.idle":"2021-05-27T16:03:19.878933Z","shell.execute_reply.started":"2021-05-27T15:59:50.816177Z","shell.execute_reply":"2021-05-27T16:03:19.877884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls /kaggle/working/","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:03:19.882686Z","iopub.execute_input":"2021-05-27T16:03:19.882967Z","iopub.status.idle":"2021-05-27T16:03:20.564541Z","shell.execute_reply.started":"2021-05-27T16:03:19.882938Z","shell.execute_reply":"2021-05-27T16:03:20.563575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# embeddings_list_available = [c.strip() for c in \"\"\"glove.840B.300d/glove.840B.300d.txt  \n# GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin  \n# wiki-news-300d-1M/wiki-news-300d-1M.vec  \n# paragram_300_sl999/README.txt  \n# paragram_300_sl999/paragram_300_sl999.txt \"\"\".split('\\n')]","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:03:20.568115Z","iopub.execute_input":"2021-05-27T16:03:20.568382Z","iopub.status.idle":"2021-05-27T16:03:22.721743Z","shell.execute_reply.started":"2021-05-27T16:03:20.568353Z","shell.execute_reply":"2021-05-27T16:03:22.720722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embeddings_unzip_path = \"\"\"  /kaggle/working/embeddings/glove.840B.300d/glove.840B.300d.txt  \n  /kaggle/working/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin  \n  /kaggle/working/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec  \n  /kaggle/working/embeddings/paragram_300_sl999/README.txt  \n  /kaggle/working/embeddings/paragram_300_sl999/paragram_300_sl999.txt \"\"\".split(\"\\n\")","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:03:22.726802Z","iopub.execute_input":"2021-05-27T16:03:22.731361Z","iopub.status.idle":"2021-05-27T16:03:25.027315Z","shell.execute_reply.started":"2021-05-27T16:03:22.731317Z","shell.execute_reply":"2021-05-27T16:03:25.026229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embeddings_list_available = [c.strip() for c in embeddings_unzip_path]","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:03:25.028798Z","iopub.execute_input":"2021-05-27T16:03:25.029382Z","iopub.status.idle":"2021-05-27T16:03:25.052488Z","shell.execute_reply.started":"2021-05-27T16:03:25.029339Z","shell.execute_reply":"2021-05-27T16:03:25.051425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embeddings_list_available","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:03:25.054405Z","iopub.execute_input":"2021-05-27T16:03:25.055058Z","iopub.status.idle":"2021-05-27T16:03:25.079399Z","shell.execute_reply.started":"2021-05-27T16:03:25.055022Z","shell.execute_reply":"2021-05-27T16:03:25.078265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embeddings_available_dict = {}\nfor idx, c in enumerate(embeddings_list_available):\n    print(f\"Embedings: {idx}. {c}\")\n    embeddings_available_dict[idx] = c","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:03:25.084362Z","iopub.execute_input":"2021-05-27T16:03:25.099088Z","iopub.status.idle":"2021-05-27T16:03:25.129868Z","shell.execute_reply.started":"2021-05-27T16:03:25.099047Z","shell.execute_reply":"2021-05-27T16:03:25.122702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embeddings_available_dict","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:03:25.131628Z","iopub.execute_input":"2021-05-27T16:03:25.132122Z","iopub.status.idle":"2021-05-27T16:03:25.148786Z","shell.execute_reply.started":"2021-05-27T16:03:25.132088Z","shell.execute_reply":"2021-05-27T16:03:25.147866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EMBEDDING_FILE = embeddings_available_dict[0]\nprint(EMBEDDING_FILE)","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:03:25.150088Z","iopub.execute_input":"2021-05-27T16:03:25.150607Z","iopub.status.idle":"2021-05-27T16:03:26.876481Z","shell.execute_reply.started":"2021-05-27T16:03:25.150555Z","shell.execute_reply":"2021-05-27T16:03:26.875441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndef get_coefs(word,*arr): \n    return word, np.asarray(arr, dtype='float32')\n\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE))","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:03:26.879844Z","iopub.execute_input":"2021-05-27T16:03:26.883674Z","iopub.status.idle":"2021-05-27T16:07:29.417739Z","shell.execute_reply.started":"2021-05-27T16:03:26.880194Z","shell.execute_reply":"2021-05-27T16:07:29.416288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embeddings_index","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2021-05-27T16:07:29.419034Z","iopub.execute_input":"2021-05-27T16:07:29.419363Z","iopub.status.idle":"2021-05-27T16:07:32.659051Z","shell.execute_reply.started":"2021-05-27T16:07:29.419327Z","shell.execute_reply":"2021-05-27T16:07:32.658200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:07:32.660701Z","iopub.execute_input":"2021-05-27T16:07:32.662206Z","iopub.status.idle":"2021-05-27T16:07:42.449059Z","shell.execute_reply.started":"2021-05-27T16:07:32.662162Z","shell.execute_reply":"2021-05-27T16:07:42.448169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport time\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport math\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, Conv1D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D\nfrom keras.models import Model\nfrom keras import initializers, regularizers, constraints, optimizers, layers","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:07:42.478926Z","iopub.execute_input":"2021-05-27T16:07:42.479216Z","iopub.status.idle":"2021-05-27T16:07:48.048917Z","shell.execute_reply.started":"2021-05-27T16:07:42.479188Z","shell.execute_reply":"2021-05-27T16:07:48.047985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## split to train and val\ntrain_df, val_df = train_test_split(train_df, test_size=0.1, random_state=2018)","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:07:48.050233Z","iopub.execute_input":"2021-05-27T16:07:48.050605Z","iopub.status.idle":"2021-05-27T16:07:48.678705Z","shell.execute_reply.started":"2021-05-27T16:07:48.050547Z","shell.execute_reply":"2021-05-27T16:07:48.677830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:07:48.680148Z","iopub.execute_input":"2021-05-27T16:07:48.680659Z","iopub.status.idle":"2021-05-27T16:07:48.698107Z","shell.execute_reply.started":"2021-05-27T16:07:48.680621Z","shell.execute_reply":"2021-05-27T16:07:48.697311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## some config values \nembed_size = 300 # how big is each word vector\nmax_features = 50000 # how many unique words to use (i.e num rows in embedding vector)\nmaxlen = 100 # max number of words in a question to use\n\n## fill up the missing values\ntrain_X = train_df[\"question_text\"].fillna(\"_na_\").values\nval_X = val_df[\"question_text\"].fillna(\"_na_\").values\ntest_X = test_df[\"question_text\"].fillna(\"_na_\").values\n\n## Tokenize the sentences\ntokenizer = Tokenizer(num_words=max_features)\ntokenizer.fit_on_texts(list(train_X))\ntrain_X = tokenizer.texts_to_sequences(train_X)\nval_X = tokenizer.texts_to_sequences(val_X)\ntest_X = tokenizer.texts_to_sequences(test_X)\n\n## Pad the sentences \ntrain_X = pad_sequences(train_X, maxlen=maxlen)\nval_X = pad_sequences(val_X, maxlen=maxlen)\ntest_X = pad_sequences(test_X, maxlen=maxlen)\n\n## Get the target values\ntrain_y = train_df['target'].values\nval_y = val_df['target'].values","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:07:48.699360Z","iopub.execute_input":"2021-05-27T16:07:48.699737Z","iopub.status.idle":"2021-05-27T16:08:48.713325Z","shell.execute_reply.started":"2021-05-27T16:07:48.699701Z","shell.execute_reply":"2021-05-27T16:08:48.712422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"question_text\"][0]","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:08:48.714660Z","iopub.execute_input":"2021-05-27T16:08:48.714994Z","iopub.status.idle":"2021-05-27T16:08:48.758225Z","shell.execute_reply.started":"2021-05-27T16:08:48.714960Z","shell.execute_reply":"2021-05-27T16:08:48.757240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X[0]","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:08:48.759532Z","iopub.execute_input":"2021-05-27T16:08:48.759893Z","iopub.status.idle":"2021-05-27T16:08:48.766857Z","shell.execute_reply.started":"2021-05-27T16:08:48.759856Z","shell.execute_reply":"2021-05-27T16:08:48.765918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"word_index = tokenizer.word_index","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:08:48.768127Z","iopub.execute_input":"2021-05-27T16:08:48.768632Z","iopub.status.idle":"2021-05-27T16:08:48.776429Z","shell.execute_reply.started":"2021-05-27T16:08:48.768594Z","shell.execute_reply":"2021-05-27T16:08:48.775654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nb_words = min(max_features, len(word_index))\nembedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:08:48.777832Z","iopub.execute_input":"2021-05-27T16:08:48.778471Z","iopub.status.idle":"2021-05-27T16:08:49.378277Z","shell.execute_reply.started":"2021-05-27T16:08:48.778433Z","shell.execute_reply":"2021-05-27T16:08:49.377341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"emb_mean#embedding_matrix.shape","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:08:49.379605Z","iopub.execute_input":"2021-05-27T16:08:49.379948Z","iopub.status.idle":"2021-05-27T16:08:49.387334Z","shell.execute_reply.started":"2021-05-27T16:08:49.379913Z","shell.execute_reply":"2021-05-27T16:08:49.385954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embedding_matrix","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:08:49.391699Z","iopub.execute_input":"2021-05-27T16:08:49.392298Z","iopub.status.idle":"2021-05-27T16:08:49.400248Z","shell.execute_reply.started":"2021-05-27T16:08:49.392257Z","shell.execute_reply":"2021-05-27T16:08:49.398895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n        ","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:08:49.402241Z","iopub.execute_input":"2021-05-27T16:08:49.402821Z","iopub.status.idle":"2021-05-27T16:08:49.560178Z","shell.execute_reply.started":"2021-05-27T16:08:49.402783Z","shell.execute_reply":"2021-05-27T16:08:49.559298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inp = Input(shape=(maxlen,))\nx = Embedding(max_features, embed_size, weights=[embedding_matrix])(inp)\nx = Bidirectional(LSTM(64, return_sequences=True))(x)\nx = GlobalMaxPool1D()(x)\nx = Dense(16, activation=\"relu\")(x)\nx = Dropout(0.1)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\nprint(model.summary())","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:08:49.561455Z","iopub.execute_input":"2021-05-27T16:08:49.561996Z","iopub.status.idle":"2021-05-27T16:08:52.562123Z","shell.execute_reply.started":"2021-05-27T16:08:49.561958Z","shell.execute_reply":"2021-05-27T16:08:52.561342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(train_X, train_y, batch_size=512, epochs=2, validation_data=(val_X, val_y))","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:08:52.564846Z","iopub.execute_input":"2021-05-27T16:08:52.565093Z","iopub.status.idle":"2021-05-27T16:30:22.748432Z","shell.execute_reply.started":"2021-05-27T16:08:52.565068Z","shell.execute_reply":"2021-05-27T16:30:22.747658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_glove_val_y = model.predict([val_X], batch_size=1024, verbose=1)\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_glove_val_y>thresh).astype(int))))","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:30:22.749817Z","iopub.execute_input":"2021-05-27T16:30:22.750181Z","iopub.status.idle":"2021-05-27T16:30:29.741680Z","shell.execute_reply.started":"2021-05-27T16:30:22.750143Z","shell.execute_reply":"2021-05-27T16:30:29.740698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_X.shape","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:34:14.814072Z","iopub.execute_input":"2021-05-27T16:34:14.814397Z","iopub.status.idle":"2021-05-27T16:34:14.821570Z","shell.execute_reply.started":"2021-05-27T16:34:14.814366Z","shell.execute_reply":"2021-05-27T16:34:14.820689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.shape","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:34:45.137910Z","iopub.execute_input":"2021-05-27T16:34:45.138246Z","iopub.status.idle":"2021-05-27T16:34:45.143750Z","shell.execute_reply.started":"2021-05-27T16:34:45.138217Z","shell.execute_reply":"2021-05-27T16:34:45.142810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub = pd.read_csv(sample_submission_filepath)","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:35:40.582693Z","iopub.execute_input":"2021-05-27T16:35:40.583027Z","iopub.status.idle":"2021-05-27T16:35:40.930395Z","shell.execute_reply.started":"2021-05-27T16:35:40.582999Z","shell.execute_reply":"2021-05-27T16:35:40.929554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_glove_test_y = model.predict([test_X], batch_size=1024, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:30:29.742947Z","iopub.execute_input":"2021-05-27T16:30:29.743302Z","iopub.status.idle":"2021-05-27T16:30:44.639329Z","shell.execute_reply.started":"2021-05-27T16:30:29.743264Z","shell.execute_reply":"2021-05-27T16:30:44.638551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub['prediction'] = (pred_glove_test_y>0.5).astype('int')","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:36:34.689468Z","iopub.execute_input":"2021-05-27T16:36:34.689812Z","iopub.status.idle":"2021-05-27T16:36:34.697283Z","shell.execute_reply.started":"2021-05-27T16:36:34.689782Z","shell.execute_reply":"2021-05-27T16:36:34.696422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2021-05-27T16:37:55.680628Z","iopub.execute_input":"2021-05-27T16:37:55.680983Z","iopub.status.idle":"2021-05-27T16:37:56.683868Z","shell.execute_reply.started":"2021-05-27T16:37:55.680927Z","shell.execute_reply":"2021-05-27T16:37:56.682993Z"},"trusted":true},"execution_count":null,"outputs":[]}]}