{"cells":[{"metadata":{"_uuid":"8f76278516670a7c801ad03bb7fcc58fbbb21524"},"cell_type":"markdown","source":"Tensorflow hub provides a very nice [universal sentence encoder module](https://tfhub.dev/google/universal-sentence-encoder/2). Unfortunately, it can only bne accessed through an online connection, which is not permitted in this competition. I am still leaving the kernel here for educational purposes."},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom time import time\nfrom tqdm import tqdm\nimport gc\n\nimport lightgbm as lgb\n\nfrom sklearn.metrics import f1_score\nfrom sklearn.model_selection import KFold\nfrom sklearn.linear_model import LogisticRegression\n\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom __future__ import absolute_import, division, print_function\nimport tensorflow as tf\nimport tensorflow_hub as hub\n\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7033aa733f4d1db08d71317fbffda98c6f9c2b2c"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv').fillna(' ')\ntest = pd.read_csv('../input/test.csv').fillna(' ')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"44a3623395c494c06e3142feca77c074f3b6f4df"},"cell_type":"code","source":"test_text = test['question_text'].values.tolist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bdc3445e349e13375cbd9809b33662597a06987d"},"cell_type":"code","source":"start_time = time()\nembed = hub.Module(\"https://tfhub.dev/google/universal-sentence-encoder/2\")\ntime() - start_time","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0e427980f0b2fe638ffbb733d6471b028f55d7dd"},"cell_type":"code","source":"embeddings = embed(test_text)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4997d24832404325d869ea0af97d3faf12bedf56"},"cell_type":"code","source":"with tf.Session() as session:\n    session.run([tf.global_variables_initializer(), tf.tables_initializer()])\n    test_embeddings = session.run(embeddings)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e56f392353d003180f412eb0923505d172d5ec41"},"cell_type":"code","source":"test_embeddings.shape","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"54247b3a020fba5001979cd5141e3cfb2053934f"},"cell_type":"markdown","source":"For train text we'll do somethign slightly different. In my previous attempts at emebdding large lists of texts in a Kaggle kernel I ran into mamory/timout issues. So I decided to break the list into smaller chunks, end embed one of them at the time. We'll do something like that here too.\n\nFirst, let's extract the training text from the train file:"},{"metadata":{"trusted":true,"_uuid":"bebd13c93e0a88fa0dd0f10a260070e3d62a2126"},"cell_type":"code","source":"train_text = train['question_text'].values.tolist()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"31815957afbc8819a5cc0cb0f62955a8393128fe"},"cell_type":"markdown","source":"We'd like to chunk it into smaller segments, each of which is approximatley the same lenght as the test text. Turns out that if we aim for about 25 chunks we can get there:"},{"metadata":{"trusted":true,"_uuid":"6c0dcc826d4738e070b771a033bb938a01c8ac46"},"cell_type":"code","source":"len(train_text)/25","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"acddcad750fb662ac72352987349f6fc46297c65"},"cell_type":"markdown","source":"Now we \"chunk\" it:"},{"metadata":{"trusted":true,"_uuid":"a88c6b0d6ece35db137d6501b3e10330f4c204ea"},"cell_type":"code","source":"train_text = [train_text[i:i + 52250] for i in range(0, len(train_text), 52250)]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e95fe0653f62444ce58f625077d8394825507d64"},"cell_type":"code","source":"len(train_text)\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5b2299f650fbf8a109b5128ee88132b7c061c0b1"},"cell_type":"markdown","source":"And now we embed each chunk individually. This takes about 15 minutes. "},{"metadata":{"trusted":true,"_uuid":"eb75d0c07717a8544ba3216133f3e18275eba451"},"cell_type":"code","source":"embeddings_train = []\nfor i in tqdm(range(25)):\n    embeddings = embed(train_text[i])\n    embeddings_train.append(embeddings)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f4a42ab8097c83a8415d5d58f19aec7b4a8661aa"},"cell_type":"code","source":"train_embeddings_all = []\nwith tf.Session() as session:\n    session.run([tf.global_variables_initializer(), tf.tables_initializer()])\n    for i in tqdm(range(25)):\n        train_embeddings = session.run(embeddings_train[i])\n        train_embeddings_all.append(train_embeddings)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e12e5c0ba1b374abaf8f81baec9db1e8bc6f0c37"},"cell_type":"code","source":"del train_text, test_text\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"993142c27bf18f1a7d604bffd48fd8b917f45e59"},"cell_type":"code","source":"train_embeddings_all = np.vstack(train_embeddings_all)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d55c029e2fd356a3b25c2a8a43c1c1b4a86c6b68"},"cell_type":"code","source":"train_embeddings_all.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7954ef91d506d959cdc11d0f52b78ffa73a7aee5"},"cell_type":"code","source":"train_target = train['target'].values\ndel train, test\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"483633577e486be1ac004c0ee423f1bfc5602845"},"cell_type":"code","source":"kf = KFold(n_splits=5, shuffle=True, random_state=43)\ntest_pred_tf = 0\noof_pred_tf = np.zeros([train_embeddings_all.shape[0],])\n\nfor i, (train_index, val_index) in tqdm(enumerate(kf.split(train_embeddings_all))):\n    x_train, x_val = train_embeddings_all[train_index,:], train_embeddings_all[val_index,:]\n    y_train, y_val = train_target[train_index], train_target[val_index]\n    classifier = LogisticRegression(class_weight = \"balanced\", C=0.5, solver='sag')\n    classifier.fit(x_train, y_train)\n    val_preds = classifier.predict_proba(x_val)[:,1]\n    preds = classifier.predict_proba(test_embeddings)[:,1]\n    test_pred_tf += 0.2*preds\n    oof_pred_tf[val_index] = val_preds","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cbf71b729a594feb005892f9a21447bee4b16bc5"},"cell_type":"code","source":"np.save('train_embeddings_all', train_embeddings_all)\nnp.save('test_embeddings', test_embeddings)\nnp.save('train_target', train_target)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fde3f258f764df9763be8431453d16c3515f5dc3"},"cell_type":"code","source":"pred_train = (oof_pred_tf > 0.8).astype(np.int)\nf1_score(train_target, pred_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e894a7b7843158c4f1c9db01054037b545c4c72e"},"cell_type":"code","source":"test = pd.read_csv('../input/test.csv').fillna(' ')\npred_test = (test_pred_tf> 0.8).astype(np.int)\nsubmission = pd.DataFrame.from_dict({'qid': test['qid']})\nsubmission['prediction'] = pred_test\nsubmission.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"358e3b97943a0fce905366d5354620545d7431a0"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}