{"cells":[{"metadata":{"_uuid":"cf9776711a4f410548c8d13fea5b144346fb08d0"},"cell_type":"markdown","source":"**引用函式庫**"},{"metadata":{"trusted":true,"_uuid":"ab8ffe395f8cb0fbd6572a8abdfff1ab2bf48f2e","scrolled":true},"cell_type":"code","source":"from keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.utils import to_categorical\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D, CuDNNLSTM\nfrom keras.models import Model\n\nimport numpy as np\nimport pandas as pd\nimport keras\nimport os\nimport re\nfrom tqdm import tqdm\ntqdm.pandas()\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1f77daeaea8154afa85c51154f6f37eeeb20f767"},"cell_type":"markdown","source":"** 文字處理**"},{"metadata":{"trusted":true,"_uuid":"6cefc71922812b8545e877af51676e1edee39f65"},"cell_type":"code","source":"## 特殊符號\ndef clean_text(x):\n    x = str(x)\n    for punct in \"/-'\":\n        x = x.replace(punct, ' ')\n    for punct in '&':\n        x = x.replace(punct, f' {punct} ')\n    for punct in '?!.,\"#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~' + '“”’':\n        x = x.replace(punct, '')\n    return x\n\n## 二位數～五位數的數字\ndef clean_numbers(x):\n    x = re.sub('[0-9]{5,}', '#####', x)\n    x = re.sub('[0-9]{4}', '####', x)\n    x = re.sub('[0-9]{3}', '###', x)\n    x = re.sub('[0-9]{2}', '##', x)\n    return x\n\n## 一些單字的統稱、過去式跟現在式統一\ndef _get_mispell(mispell_dict):\n    mispell_re = re.compile('(%s)' % '|'.join(mispell_dict.keys()))\n    return mispell_dict, mispell_re\n\nmispell_dict = {'colour':'color',\n                'centre':'center',\n                'didnt':'did not',\n                'doesnt':'does not',\n                'isnt':'is not',\n                'shouldnt':'should not',\n                'favourite':'favorite',\n                'travelling':'traveling',\n                'counselling':'counseling',\n                'theatre':'theater',\n                'cancelled':'canceled',\n                'labour':'labor',\n                'organisation':'organization',\n                'wwii':'world war 2',\n                'citicise':'criticize',\n                'instagram': 'social medium',\n                'whatsapp': 'social medium',\n                'snapchat': 'social medium'\n                }\nmispellings, mispellings_re = _get_mispell(mispell_dict)\n\ndef replace_typical_misspell(text):\n    def replace(match):\n        return mispellings[match.group(0)]\n    return mispellings_re.sub(replace, text)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8ae444c3561846cbe066ee89c73dc35d82022299"},"cell_type":"markdown","source":"## Load train & test data"},{"metadata":{"trusted":true,"_uuid":"fa3b83c59fa8cb8ba795f9687147f4792986545e"},"cell_type":"code","source":"test = pd.read_csv('../input/test.csv')\ntrain = pd.read_csv('../input/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e1c4a4737969bcfe1a0407d5018bd759c7844d5b"},"cell_type":"code","source":"print(test.shape)\nprint(train.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"069a32e9f721b11a590f588d3dcb09084b9613be"},"cell_type":"code","source":"MAX_NB_WORDS = 50000\nEMBEDDING_DIM = 300\nMAX_SEQUENCE_LENGTH = 50\ny_train = train['target'].values\n#train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"03ca5785b89e1d2d58f1a72335c46a588e7a3062"},"cell_type":"code","source":"# train[train['target']==1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0b8bc82a647a5c540b2601ef1e9e789ce3178bd2"},"cell_type":"code","source":"# train.loc[0]['question_text']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6c8978157436bd83d044053845f5055bac834107"},"cell_type":"code","source":"# train['question_text'].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0273e595ca940f2b9f7be2d6d120b62077b4b176","scrolled":true},"cell_type":"code","source":"# Clean the text\ntrain[\"question_text\"] = train[\"question_text\"].progress_apply(lambda x: clean_text(x))\ntest[\"question_text\"] = test[\"question_text\"].progress_apply(lambda x: clean_text(x))\n    \n# Clean numbers\ntrain[\"question_text\"] = train[\"question_text\"].progress_apply(lambda x: clean_numbers(x))\ntest[\"question_text\"] = test[\"question_text\"].progress_apply(lambda x: clean_numbers(x))\n    \n# Clean speelings\ntrain[\"question_text\"] = train[\"question_text\"].progress_apply(lambda x: replace_typical_misspell(x))\ntest[\"question_text\"] = test[\"question_text\"].progress_apply(lambda x: replace_typical_misspell(x))\n    \n# fill up the missing values\n# train_X = train[\"question_text\"].fillna(\"_##_\").values\n# test_X = test[\"question_text\"].fillna(\"_##_\").values","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"29d9b7643f0fb33d7dc03a55b6f649dbe0990d5b"},"cell_type":"markdown","source":"## Load Eebedding data (glove)"},{"metadata":{"trusted":true,"_uuid":"b105928c652e3c6d1d40ef7b77253980bb04ee72","scrolled":false},"cell_type":"code","source":"# load embeddings \nembeddings_glove = {}\nwith open ('../input/embeddings/glove.840B.300d/glove.840B.300d.txt', 'r', encoding='UTF-8') as f:\n    for line in f.readlines():\n        line = line.split(\" \")\n        key = line[0]\n        values = np.asarray(line[1:], dtype='float32')\n        embeddings_glove[key] = values\n#         print(key)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"37656f49f04c032181b555729c2893cf6039742d","scrolled":true},"cell_type":"code","source":"embeddings_glove['a']","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"57c7c119aba4d935b048987e85281439d418558b"},"cell_type":"markdown","source":"## Create Token & Pad sequences"},{"metadata":{"trusted":true,"_uuid":"50406dd2113fa6c86e580450c31a20d0e318f8d6"},"cell_type":"code","source":"# create token 建token表\ntoken = Tokenizer(num_words=MAX_NB_WORDS)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"44b5be9bdf800b676704b9ba8b5f3dbf10f0d1fd"},"cell_type":"code","source":"token.fit_on_texts(list(train['question_text'].values))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3477b50940a75df76aa64eba314c64188608e3c6","scrolled":true},"cell_type":"code","source":"token.word_index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3d9c917580af41aa9d87cf9f7cb1a644474ead05"},"cell_type":"code","source":"# 利用建好的token表，將問題字串轉換\nx_train = token.texts_to_sequences(train['question_text'].fillna(\"_##_\").values)\nx_test = token.texts_to_sequences(test['question_text'].fillna(\"_##_\").values)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6616709aa16cae99f13869e3973a13d216f38ddc"},"cell_type":"code","source":"print(train['question_text'].values[0])\nprint(x_train[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8096efa8129661095e4fdc88d840d92405f8a802"},"cell_type":"code","source":"# 將問題字串截長補短\nx_train = pad_sequences(x_train, maxlen=MAX_SEQUENCE_LENGTH)\nx_test = pad_sequences(x_test, maxlen=MAX_SEQUENCE_LENGTH)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aec80651f2156ea92e1d388c3078a52b69b2040a"},"cell_type":"code","source":"print(x_train[0])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"75c1aa9a285124afe96807ba440bad00231b9607"},"cell_type":"markdown","source":"## Get input vector"},{"metadata":{"trusted":true,"_uuid":"e1fd841712a1bc049a994fe024ff3ad05cdb0511"},"cell_type":"code","source":"num_words = min(MAX_NB_WORDS, len(token.word_index))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4611678309d2ecc60d5d6f72c3cab234665f96b5"},"cell_type":"code","source":"num_words","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"62816ac9ad0e36c0a99f04db4e7f7fb0a71a9b25"},"cell_type":"code","source":"embedding_matrix = np.zeros((num_words, EMBEDDING_DIM))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"520a7630dfd26ff07cbd3e7433655a651074a124","scrolled":true},"cell_type":"code","source":"# print(embedding_matrix)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"281362d426c8f5c3ed89cb9116726b3fa6461e7c"},"cell_type":"code","source":"embedding_matrix.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e332ea2322326c75122c4f6f49cd7e4ce9705451"},"cell_type":"code","source":"for word, i in token.word_index.items():\n    if i >= MAX_NB_WORDS:\n        continue\n    embedding_vector = embeddings_glove.get(word)\n    if embedding_vector is not None:\n        # words not found in embedding index will be all-zeros.\n        embedding_matrix[i] = embedding_vector","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1ebefbcabd762ce5a87fe0f09926dc8b0912a80f"},"cell_type":"code","source":"embedding_matrix","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"10f9503947bbfd5a70c910da79fbba166397d52a"},"cell_type":"markdown","source":"## Create model"},{"metadata":{"trusted":true,"_uuid":"d26e44d0bb19ac7128f5f8bdb84e53757d9efae3"},"cell_type":"code","source":"embedding_layer = Embedding(num_words,\n                            EMBEDDING_DIM,\n                            weights=[embedding_matrix],\n                            input_length=MAX_SEQUENCE_LENGTH,\n                            trainable=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"646188e149ac19ac2f3b05481f59083b47bec830","scrolled":true},"cell_type":"code","source":"inp = Input(shape=(MAX_SEQUENCE_LENGTH,))\nx = embedding_layer(inp)\nx = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\nx = Bidirectional(CuDNNGRU(64))(x)\nx = Dropout(0.2)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n\nprint(model.summary())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c069ea60b35151ba99f59b454a116f00bdbed7e9"},"cell_type":"code","source":"model.fit(x_train, y_train,\n          batch_size=128,\n          epochs=8, \n          validation_split=0.1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"92efa13a61f330a8be9a15e6b022006d5fee4b08"},"cell_type":"code","source":"glove_test_y  = model.predict(x_test, batch_size=128, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"019546006be79ebf12712178c0a4dfc3bf4d5815"},"cell_type":"code","source":"# for i in glove_test_y:\n#     print(i)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b615a03c8257dadf05ff3814c438df795b888834"},"cell_type":"code","source":"test_y = (glove_test_y>0.4).astype(int)\nout = pd.DataFrame({\"qid\":test[\"qid\"].values})\nout['prediction'] = test_y\nout.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}