{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\n\nimport os\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport math\nfrom sklearn.model_selection import train_test_split\nfrom keras.preprocessing import text","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.status.busy":"2021-10-29T14:23:33.377194Z","iopub.execute_input":"2021-10-29T14:23:33.377513Z","iopub.status.idle":"2021-10-29T14:23:34.175434Z","shell.execute_reply.started":"2021-10-29T14:23:33.377454Z","shell.execute_reply":"2021-10-29T14:23:34.174659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Setup","metadata":{"_uuid":"21f85dfd5b9bf3d8b27ba29149d52253e5d64049"}},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/quora-insincere-questions-classification/train.csv\")\ntrain_df, val_df = train_test_split(train_df, test_size=0.1)","metadata":{"_uuid":"78578eab64a477d0a5ad6b1c917ae154868a44df","execution":{"iopub.status.busy":"2021-10-29T14:23:34.178368Z","iopub.execute_input":"2021-10-29T14:23:34.178895Z","iopub.status.idle":"2021-10-29T14:23:38.604802Z","shell.execute_reply.started":"2021-10-29T14:23:34.178841Z","shell.execute_reply":"2021-10-29T14:23:38.604047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!unzip ../input/quora-insincere-questions-classification/embeddings.zip -d \"./\"","metadata":{"execution":{"iopub.status.busy":"2021-10-29T14:23:38.609306Z","iopub.execute_input":"2021-10-29T14:23:38.609560Z","iopub.status.idle":"2021-10-29T14:27:09.234649Z","shell.execute_reply.started":"2021-10-29T14:23:38.609515Z","shell.execute_reply":"2021-10-29T14:27:09.233829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir(\"./glove.840B.300d\")","metadata":{"execution":{"iopub.status.busy":"2021-10-29T14:27:09.237294Z","iopub.execute_input":"2021-10-29T14:27:09.237684Z","iopub.status.idle":"2021-10-29T14:27:09.248150Z","shell.execute_reply.started":"2021-10-29T14:27:09.237560Z","shell.execute_reply":"2021-10-29T14:27:09.247443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# embdedding setup../input/embeddings/embeddings/glove.840B.300d\n# Source https://blog.keras.io/using-pre-trained-word-embeddings-in-a-keras-model.html\nembeddings_index = {}\nf = open('./glove.840B.300d/glove.840B.300d.txt')\nfor line in tqdm(f):\n    values = line.split(\" \")\n    word = values[0]\n    coefs = np.asarray(values[1:], dtype='float32')\n    embeddings_index[word] = coefs\nf.close()\n\nprint('Found %s word vectors.' % len(embeddings_index))","metadata":{"_uuid":"92ffbf2ef35d2dc5863ee27c61eadd1869ca5440","execution":{"iopub.status.busy":"2021-10-29T14:27:09.250088Z","iopub.execute_input":"2021-10-29T14:27:09.250351Z","iopub.status.idle":"2021-10-29T14:31:44.317289Z","shell.execute_reply.started":"2021-10-29T14:27:09.250295Z","shell.execute_reply":"2021-10-29T14:31:44.316452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert values to embeddings\ndef text_to_array(text):\n    empyt_emb = np.zeros(300)\n    text = text[:-1].split()[:30]\n    embeds = [embeddings_index.get(x, empyt_emb) for x in text]\n    embeds+= [empyt_emb] * (30 - len(embeds))\n    return np.array(embeds)\n\n# train_vects = [text_to_array(X_text) for X_text in tqdm(train_df[\"question_text\"])]\nval_vects = np.array([text_to_array(X_text) for X_text in tqdm(val_df[\"question_text\"][:3000])])\nval_y = np.array(val_df[\"target\"][:3000])\n","metadata":{"_uuid":"9cc8b0bcd285225ce651d024f506615d4656b9f7","execution":{"iopub.status.busy":"2021-10-29T14:31:44.318391Z","iopub.execute_input":"2021-10-29T14:31:44.318688Z","iopub.status.idle":"2021-10-29T14:31:44.721227Z","shell.execute_reply.started":"2021-10-29T14:31:44.318639Z","shell.execute_reply":"2021-10-29T14:31:44.720465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data providers\nbatch_size = 128\n\ndef batch_gen(train_df):\n    n_batches = math.ceil(len(train_df) / batch_size)\n    while True: \n        train_df = train_df.sample(frac=1.)  # Shuffle the data.\n        for i in range(n_batches):\n            texts = train_df.iloc[i*batch_size:(i+1)*batch_size, 1]\n            text_arr = np.array([text_to_array(text) for text in texts])\n            yield text_arr, np.array(train_df[\"target\"][i*batch_size:(i+1)*batch_size])\n","metadata":{"_uuid":"0c950448e0717eaebb920d93cfdc6b4561e21853","execution":{"iopub.status.busy":"2021-10-29T14:31:44.722138Z","iopub.execute_input":"2021-10-29T14:31:44.722378Z","iopub.status.idle":"2021-10-29T14:31:44.729759Z","shell.execute_reply.started":"2021-10-29T14:31:44.722323Z","shell.execute_reply":"2021-10-29T14:31:44.728661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{"_uuid":"706c0224b6112e5a8f00ad25f35e90fbb9519a5f"}},{"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers import CuDNNLSTM, Dense, Bidirectional\nimport tensorflow as tf\nfrom keras.layers import LSTM\nfrom keras.layers import TimeDistributed\nfrom keras.layers import Bidirectional","metadata":{"_uuid":"798c303ec834fb530a60a1e590cfbd9a86f93fde","execution":{"iopub.status.busy":"2021-10-29T14:31:44.730949Z","iopub.execute_input":"2021-10-29T14:31:44.731765Z","iopub.status.idle":"2021-10-29T14:31:44.740506Z","shell.execute_reply.started":"2021-10-29T14:31:44.731418Z","shell.execute_reply":"2021-10-29T14:31:44.739545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model = Sequential()\n# model.add(Bidirectional(LSTM(20, return_sequences=True), input_shape=(30, 300)))\n# model.add(Dense(1, activation=\"sigmoid\"))\n# model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['acc'])","metadata":{"execution":{"iopub.status.busy":"2021-10-29T14:31:44.741493Z","iopub.execute_input":"2021-10-29T14:31:44.741741Z","iopub.status.idle":"2021-10-29T14:31:44.753795Z","shell.execute_reply.started":"2021-10-29T14:31:44.741687Z","shell.execute_reply":"2021-10-29T14:31:44.752981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential()\nmodel.add(Bidirectional(LSTM(64, return_sequences=True),input_shape=(30, 300)))\nmodel.add(Bidirectional(LSTM(64)))\nmodel.add(Dense(1, activation=\"sigmoid\"))\n\nmodel.compile(loss='binary_crossentropy',\n              optimizer='adam',\n              metrics=['accuracy'])","metadata":{"_uuid":"9ad418c95b31ab5691d49b72c7c9622ef9ea42cf","execution":{"iopub.status.busy":"2021-10-29T14:31:44.754584Z","iopub.execute_input":"2021-10-29T14:31:44.754797Z","iopub.status.idle":"2021-10-29T14:31:46.027463Z","shell.execute_reply.started":"2021-10-29T14:31:44.754754Z","shell.execute_reply":"2021-10-29T14:31:46.026624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mg = batch_gen(train_df)\nmodel.fit_generator(mg, epochs=10,\n                    steps_per_epoch=1000,\n                    validation_data=(val_vects, val_y),\n                    verbose=True)","metadata":{"_uuid":"3f0b703ded691293450beb4ddf2d903d0e08c737","execution":{"iopub.status.busy":"2021-10-29T14:33:14.483903Z","iopub.execute_input":"2021-10-29T14:33:14.484419Z","iopub.status.idle":"2021-10-29T14:49:58.579724Z","shell.execute_reply.started":"2021-10-29T14:33:14.484195Z","shell.execute_reply":"2021-10-29T14:49:58.579031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference","metadata":{"_uuid":"d6cdda0e301be0b84e826e29f0f3a85c41aa6da9"}},{"cell_type":"code","source":"model.save(\"BLSTM.h5\")","metadata":{"execution":{"iopub.status.busy":"2021-10-29T14:58:41.653519Z","iopub.execute_input":"2021-10-29T14:58:41.653828Z","iopub.status.idle":"2021-10-29T14:58:42.065104Z","shell.execute_reply.started":"2021-10-29T14:58:41.653773Z","shell.execute_reply":"2021-10-29T14:58:42.064276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install numpy","metadata":{"execution":{"iopub.status.busy":"2021-10-29T15:03:17.710233Z","iopub.execute_input":"2021-10-29T15:03:17.710582Z","iopub.status.idle":"2021-10-29T15:03:23.488576Z","shell.execute_reply.started":"2021-10-29T15:03:17.710527Z","shell.execute_reply":"2021-10-29T15:03:23.487521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tensorflow>=2.1.0","metadata":{"execution":{"iopub.status.busy":"2021-10-29T15:02:06.515057Z","iopub.execute_input":"2021-10-29T15:02:06.515508Z","iopub.status.idle":"2021-10-29T15:02:12.865126Z","shell.execute_reply.started":"2021-10-29T15:02:06.515309Z","shell.execute_reply":"2021-10-29T15:02:12.864323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(tf.__version__)","metadata":{"execution":{"iopub.status.busy":"2021-10-29T15:01:16.670564Z","iopub.execute_input":"2021-10-29T15:01:16.673193Z","iopub.status.idle":"2021-10-29T15:01:16.681466Z","shell.execute_reply.started":"2021-10-29T15:01:16.673134Z","shell.execute_reply":"2021-10-29T15:01:16.679554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflowjs as tfjs\ntfjs.converters.save_keras_model(model, \"tfjs_model\")","metadata":{"execution":{"iopub.status.busy":"2021-10-29T15:00:00.707577Z","iopub.execute_input":"2021-10-29T15:00:00.707878Z","iopub.status.idle":"2021-10-29T15:00:00.730373Z","shell.execute_reply.started":"2021-10-29T15:00:00.707825Z","shell.execute_reply":"2021-10-29T15:00:00.729374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prediction part\nbatch_size = 256\ndef batch_gen(test_df):\n    n_batches = math.ceil(len(test_df) / batch_size)\n    for i in range(n_batches):\n        texts = test_df.iloc[i*batch_size:(i+1)*batch_size, 1]\n        text_arr = np.array([text_to_array(text) for text in texts])\n        yield text_arr\n\ntest_df = pd.read_csv(\"../input/quora-insincere-questions-classification/test.csv\")\n\nall_preds = []\nfor x in tqdm(batch_gen(test_df)):\n    all_preds.extend(model.predict(x).flatten())","metadata":{"_uuid":"b58cd95254f41e5002a17de0c3feab54a5fc3c67","execution":{"iopub.status.busy":"2021-10-29T14:32:59.108049Z","iopub.status.idle":"2021-10-29T14:32:59.108499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_te = (np.array(all_preds) > 0.5).astype(np.int)\n\nsubmit_df = pd.DataFrame({\"qid\": test_df[\"qid\"], \"prediction\": y_te})\nsubmit_df.to_csv(\"submission.csv\", index=False)","metadata":{"_uuid":"3e6ed54def110c881f401c6f8a844752baedcfbd","execution":{"iopub.status.busy":"2021-10-29T14:32:59.109219Z","iopub.status.idle":"2021-10-29T14:32:59.109663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import tensorflowjs as tfjs\n# tfjs.converters.save_keras_model(model, tfjs_target_dir)","metadata":{"execution":{"iopub.status.busy":"2021-10-29T14:32:59.110364Z","iopub.status.idle":"2021-10-29T14:32:59.110828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}