{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import zipfile","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:39:25.674332Z","iopub.execute_input":"2022-05-04T08:39:25.674942Z","iopub.status.idle":"2022-05-04T08:39:25.680624Z","shell.execute_reply.started":"2022-05-04T08:39:25.674898Z","shell.execute_reply":"2022-05-04T08:39:25.678294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"zp = zipfile.ZipFile('../input/quora-insincere-questions-classification/embeddings.zip')\nzp.extract('GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin')","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:39:26.408281Z","iopub.execute_input":"2022-05-04T08:39:26.408707Z","iopub.status.idle":"2022-05-04T08:40:22.538142Z","shell.execute_reply.started":"2022-05-04T08:39:26.408644Z","shell.execute_reply":"2022-05-04T08:40:22.537132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:41:14.043929Z","iopub.execute_input":"2022-05-04T08:41:14.044523Z","iopub.status.idle":"2022-05-04T08:41:20.060546Z","shell.execute_reply.started":"2022-05-04T08:41:14.044488Z","shell.execute_reply":"2022-05-04T08:41:20.059457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read the data\ntrain_df = pd.read_csv('../input/quora-insincere-questions-classification/train.csv')\ntexts = train_df['question_text']\nlabels = train_df['target']\n\ntrain_x, valid_x, train_y, valid_y = train_test_split(\n    texts, \n    labels,\n    test_size=0.2,\n    stratify=labels)","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:41:35.638030Z","iopub.execute_input":"2022-05-04T08:41:35.638358Z","iopub.status.idle":"2022-05-04T08:41:39.027553Z","shell.execute_reply.started":"2022-05-04T08:41:35.638326Z","shell.execute_reply":"2022-05-04T08:41:39.026543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from gensim.models import KeyedVectors\nfrom tensorflow.keras import layers\n\n# Read the glove embeddings\ndef get_glove_vecs():\n    glove_vecs = KeyedVectors.load_word2vec_format(\n        './GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin', \n        binary=True,\n        limit=500000\n    )\n    # Construct vocab\n    vocab = list(glove_vecs.key_to_index.keys())\n    return glove_vecs.vectors, vocab\n\nglove_vecs, vocab = get_glove_vecs()\nprint(glove_vecs.shape)\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:41:46.371307Z","iopub.execute_input":"2022-05-04T08:41:46.371604Z","iopub.status.idle":"2022-05-04T08:41:56.286256Z","shell.execute_reply.started":"2022-05-04T08:41:46.371575Z","shell.execute_reply":"2022-05-04T08:41:56.285320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoder = layers.TextVectorization(\n    split='whitespace',\n    vocabulary=vocab,\n    output_sequence_length=50,\n    standardize=lambda tx: tf.strings.lower(tx),\n    output_mode='int'\n)","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:42:01.201162Z","iopub.execute_input":"2022-05-04T08:42:01.201745Z","iopub.status.idle":"2022-05-04T08:42:09.515592Z","shell.execute_reply.started":"2022-05-04T08:42:01.201710Z","shell.execute_reply":"2022-05-04T08:42:09.514574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 1000\ntrain_dataset = tf.data.Dataset.from_tensor_slices((train_x, train_y)).batch(batch_size)\nvalid_dataset = tf.data.Dataset.from_tensor_slices((valid_x, valid_y)).batch(batch_size)","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:42:13.783392Z","iopub.execute_input":"2022-05-04T08:42:13.784205Z","iopub.status.idle":"2022-05-04T08:42:14.203898Z","shell.execute_reply.started":"2022-05-04T08:42:13.784171Z","shell.execute_reply":"2022-05-04T08:42:14.202960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"glove_vecs = np.vstack([np.zeros(shape=(2, glove_vecs.shape[1])), glove_vecs])\nglove_vecs.shape\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:42:15.702323Z","iopub.execute_input":"2022-05-04T08:42:15.702604Z","iopub.status.idle":"2022-05-04T08:42:17.407944Z","shell.execute_reply.started":"2022-05-04T08:42:15.702573Z","shell.execute_reply":"2022-05-04T08:42:17.407006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_vec_dataset = train_dataset.map(lambda x, y: (encoder(x), y)).prefetch(10)\nvalid_vec_dataset = valid_dataset.map(lambda x, y: (encoder(x), y)).prefetch(10)","metadata":{"execution":{"iopub.status.busy":"2022-05-04T08:47:41.193910Z","iopub.execute_input":"2022-05-04T08:47:41.194732Z","iopub.status.idle":"2022-05-04T08:47:42.749247Z","shell.execute_reply.started":"2022-05-04T08:47:41.194695Z","shell.execute_reply":"2022-05-04T08:47:42.748121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Construct the model\nfrom tensorflow.keras import Model, Input\n\ninputs = Input(shape=(None,), dtype='int64')\n# x = encoder(inputs)\nx = layers.Embedding(\n    glove_vecs.shape[0],\n    glove_vecs.shape[1],\n    input_length=50,\n    weights=[glove_vecs],\n    trainable=False\n)(inputs)\nx = layers.Bidirectional(\n    layers.LSTM(128)\n)(x)\noutputs = layers.Dense(1, activation='sigmoid')(x)\nmodel = Model(inputs, outputs)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-05-04T09:00:34.845787Z","iopub.execute_input":"2022-05-04T09:00:34.846817Z","iopub.status.idle":"2022-05-04T09:00:36.381985Z","shell.execute_reply.started":"2022-05-04T09:00:34.846770Z","shell.execute_reply":"2022-05-04T09:00:36.380806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(\n    optimizer='adam',\n    loss='binary_crossentropy',\n    metrics=['accuracy']\n)","metadata":{"execution":{"iopub.status.busy":"2022-05-04T09:00:43.999332Z","iopub.execute_input":"2022-05-04T09:00:43.999654Z","iopub.status.idle":"2022-05-04T09:00:44.011721Z","shell.execute_reply.started":"2022-05-04T09:00:43.999622Z","shell.execute_reply":"2022-05-04T09:00:44.010556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(train_vec_dataset, validation_data=valid_vec_dataset, epochs=6)","metadata":{"execution":{"iopub.status.busy":"2022-05-04T09:00:48.220777Z","iopub.execute_input":"2022-05-04T09:00:48.221085Z","iopub.status.idle":"2022-05-04T09:07:46.843300Z","shell.execute_reply.started":"2022-05-04T09:00:48.221034Z","shell.execute_reply":"2022-05-04T09:07:46.842341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Obtain the optimal cuttoff on the validation dataset\nfrom sklearn.metrics import f1_score\n\nvalid_predictions = model.predict(valid_vec_dataset.map(lambda x, y: x))\n\nfor threshold in np.arange(0.1, 0.6, 0.1):\n    print(f'F1 Score at thresh {threshold}: {f1_score(valid_y, (valid_predictions > threshold)*1)}')","metadata":{"execution":{"iopub.status.busy":"2022-05-04T09:15:58.652318Z","iopub.execute_input":"2022-05-04T09:15:58.652825Z","iopub.status.idle":"2022-05-04T09:16:02.628038Z","shell.execute_reply.started":"2022-05-04T09:15:58.652777Z","shell.execute_reply":"2022-05-04T09:16:02.627118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Obtain the predictions for the test dataset\ntest_df = pd.read_csv('../input/quora-insincere-questions-classification/test.csv')\n\ntest_dataset = tf.data.Dataset.from_tensor_slices(\n    test_df['question_text']).batch(\n    100).map(lambda x: encoder(x))\n\ntest_predictions = (model.predict(test_dataset) > 0.3)*1\ntest_df['prediction'] = test_predictions\n\n# Save the results and submit\ntest_df[['qid', 'prediction']].to_csv('submission.csv', index=None)","metadata":{"execution":{"iopub.status.busy":"2022-05-04T09:18:05.675288Z","iopub.execute_input":"2022-05-04T09:18:05.675565Z","iopub.status.idle":"2022-05-04T09:18:06.398438Z","shell.execute_reply.started":"2022-05-04T09:18:05.675536Z","shell.execute_reply":"2022-05-04T09:18:06.397397Z"},"trusted":true},"execution_count":null,"outputs":[]}]}