{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nprint(os.listdir(\"../input\"))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport matplotlib.pyplot as plt\nplt.style.use(\"ggplot\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5f24aac97b644fda580bd11b1633b5d6530364c0"},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv')\ntest = pd.read_csv('../input/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f1748f387b0ffed6fbeb447354bc8e1f15c69e24"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"90f471b352120fb4ef46c1d82212b965c8eee8e7"},"cell_type":"code","source":"print('Processing text dataset')\nfrom nltk.tokenize import WordPunctTokenizer\nfrom collections import Counter\nfrom string import punctuation, ascii_lowercase\nimport regex as re\nfrom tqdm import tqdm","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"92634758d5b0ff5996355d136a102f1511f84d84"},"cell_type":"code","source":"# replace urls\nre_url = re.compile(r\"((http|https)\\:\\/\\/)?[a-zA-Z0-9\\.\\/\\?\\:@\\-_=#]+\\\n                    .([a-zA-Z]){2,6}([a-zA-Z0-9\\.\\&\\/\\?\\:@\\-_=#])*\",\n                    re.MULTILINE|re.UNICODE)\n# replace ips\nre_ip = re.compile(\"\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}\")\n\n# setup tokenizer\ntokenizer = WordPunctTokenizer()\n\nvocab = Counter()\n\ndef text_to_wordlist(text, lower=False):\n    # replace URLs\n    text = re_url.sub(\"URL\", text)\n    \n    # replace IPs\n    text = re_ip.sub(\"IPADDRESS\", text)\n    \n    # Tokenize\n    text = tokenizer.tokenize(text)\n    \n    # optional: lower case\n    if lower:\n        text = [t.lower() for t in text]\n    \n    # Return a list of words\n    vocab.update(text)\n    return text\n\ndef process_comments(list_sentences, lower=False):\n    comments = []\n    for text in tqdm(list_sentences):\n        txt = text_to_wordlist(text, lower=lower)\n        comments.append(txt)\n    return comments\n\n\nlist_sentences_train = list(train[\"question_text\"].fillna(\"NAN_WORD\").values)\nlist_sentences_test = list(test[\"question_text\"].fillna(\"NAN_WORD\").values)\n\ncomments = process_comments(list_sentences_train + list_sentences_test, lower=True)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c0beeeec36baa7f412322db1e74763d4620294be"},"cell_type":"code","source":"print(\"The vocabulary contains {} unique tokens\".format(len(vocab)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"52150e9412976341ce6ee724b287aa5df3bdcb84"},"cell_type":"code","source":"from gensim.models import Word2Vec","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0c35c3f240f216046e527eb151f7464cdc64703a"},"cell_type":"code","source":"model = Word2Vec(comments, size=100, window=5, min_count=3, workers=16, sg=0, negative=5)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a469196f000c92e0f031c5c5d814b4755573ff85"},"cell_type":"code","source":"word_vectors = model.wv\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"515f3858902e8924471317b6cff3b78c34328aa5"},"cell_type":"code","source":"print(\"Number of word vectors: {}\".format(len(word_vectors.vocab)))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"b9d16a74f50264b4f8bed4ff2f59c380d6b25a6a"},"cell_type":"code","source":"model.wv.most_similar_cosmul(positive=['woman', 'king'], negative=['man'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6f6db8f95612df12cbac08bcc7f74ba179c3cdc0"},"cell_type":"code","source":"#Initialize The Embeddings In Keras\n\nMAX_NB_WORDS = len(word_vectors.vocab) #52199\nMAX_SEQUENCE_LENGTH = 20","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3bf0579c53887162b19000394fb6ca9da568fb56"},"cell_type":"code","source":"from keras.preprocessing.sequence import pad_sequences\n\nword_index = {t[0]: i+1 for i,t in enumerate(vocab.most_common(MAX_NB_WORDS))}\nsequences = [[word_index.get(t, 0) for t in comment]\n             for comment in comments[:len(list_sentences_train)]]\ntest_sequences = [[word_index.get(t, 0)  for t in comment] \n                  for comment in comments[len(list_sentences_train):]]\n# word index -> word to number dictionary\n#sequence -> array of words to array of numbered indexes\n# pad\ndata = pad_sequences(sequences, maxlen=MAX_SEQUENCE_LENGTH, \n                     padding=\"pre\", truncating=\"post\")\ny = train['target'].values\nprint('Shape of data tensor:', data.shape)\nprint('Shape of label tensor:', y.shape)\n\ntest_data = pad_sequences(test_sequences, maxlen=MAX_SEQUENCE_LENGTH, padding=\"pre\",\n                          truncating=\"post\")\nprint('Shape of test_data tensor:', test_data.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"367d482fd62036b18b1e5045454fb047a6839ffc"},"cell_type":"code","source":"#create the embedding matrix\n\nWV_DIM = 100\nnb_words = min(MAX_NB_WORDS, len(word_vectors.vocab))\n# we initialize the matrix with random numbers\nwv_matrix = (np.random.rand(nb_words, WV_DIM) - 0.5) / 5.0\nfor word, i in word_index.items():\n    if i >= MAX_NB_WORDS:\n        continue\n    try:\n        embedding_vector = word_vectors[word]\n        # words not found in embedding index will be all-zeros.\n        wv_matrix[i] = embedding_vector\n    except:\n        pass        ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5148bf0d768843a55703424b2b1ce470c73149a3"},"cell_type":"code","source":"#Setup The Comment Classifier\n\nfrom keras.layers import Dense, Input, CuDNNLSTM, Embedding, Dropout,SpatialDropout1D, Bidirectional\nfrom keras.models import Model\nfrom keras.optimizers import Adam\nfrom keras.layers.normalization import BatchNormalization","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2b060247bfd330e777160e9af64ed6b078b0f95f","scrolled":true},"cell_type":"code","source":"wv_matrix.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"241b34af9ae4b9a7afd9e15c9ed01e4130f78c5d"},"cell_type":"code","source":"from gensim.models import KeyedVectors\nnews_path = '../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin'\nembeddings_index = KeyedVectors.load_word2vec_format(news_path, binary=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0589834a274b9ad5b5868ba948c1dfb4f302769a"},"cell_type":"code","source":"#create the embedding matrix\n\nWV_DIM = 300\nnb_words = min(MAX_NB_WORDS, len(word_vectors.vocab))\n# we initialize the matrix with random numbers\nwv_matrix_glove = (np.random.rand(nb_words, WV_DIM) - 0.5) / 5.0\nfor word, i in word_index.items():\n    if i >= MAX_NB_WORDS:\n        continue\n    try:\n        #embedding_vector = word_vectors[word]\n        # words not found in embedding index will be all-zeros.\n        wv_matrix_glove[i] = embeddings_index.get_vector(word)\n    except:\n        pass        ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c8f24fc5ca3743d6998efabdaee18cd45728b038"},"cell_type":"code","source":"wv_matrix = np.concatenate((wv_matrix,wv_matrix_glove), axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":false,"_uuid":"4d6b95e8c2f6a3f1f13d40b3488f596f976a25cb"},"cell_type":"code","source":"WV_DIM = 400\nwv_layer = Embedding(nb_words,\n                     WV_DIM,\n                     mask_zero=False,\n                     weights=[wv_matrix],\n                     input_length=MAX_SEQUENCE_LENGTH,\n                     trainable=False)\n# max words = 52199,vector dim = 100,words(52199)*vectors(100),200,\n# Inputs\ncomment_input = Input(shape=(MAX_SEQUENCE_LENGTH,), dtype='int32')\n\nembedded_sequences = wv_layer(comment_input)\n# biGRU\nembedded_sequences = SpatialDropout1D(0.2)(embedded_sequences)\nx = Bidirectional(CuDNNLSTM(64, return_sequences=False))(embedded_sequences)\n# Output\nx = Dropout(0.2)(x)\nx = BatchNormalization()(x)\npreds = Dense(1, activation='sigmoid')(x)\n\n# build the model\nmodel = Model(inputs=[comment_input], outputs=preds)\nmodel.compile(loss = 'binary_crossentropy', optimizer = 'adam', metrics = ['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"27fb1e91a77fe4ddfb4f61d928c7fc5b64e84da4"},"cell_type":"code","source":"hist = model.fit([data], y, validation_split=0.1,\n                 epochs=5, batch_size=256, shuffle=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a1b9503f5fea07a415aa8aa913e169eefe863262"},"cell_type":"code","source":"y_valpred = model.predict(data)\nf1 = 0\nthreshold = 0\nfrom sklearn import metrics\nfor thresh in np.arange(0.1, 0.501,0.01):\n    f1score = np.round(metrics.f1_score(train.target, (y_valpred>thresh).astype(int)), 4)\n    thresh = np.round(thresh,2)\n    print('F1 score for threshold {} : {}'.format(thresh, f1score))\n    if f1score > f1:\n        f1 = f1score\n        threshold = thresh\n        print('In {} : {}'.format(threshold,f1score))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ae4f79456ff1e7e2cee2b4d21f105336f7d2c331"},"cell_type":"code","source":"y_valpred_test = model.predict(test_data)\ny_test = (y_valpred_test[:,0] > threshold).astype(np.int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1714fc480738a36edd9f5c322e9710f2d60fc8e1"},"cell_type":"code","source":"submit_df = pd.DataFrame({\"qid\": test[\"qid\"], \"prediction\": y_test})\nsubmit_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"85a4c792dc3f626ee5db15fc82fa7bd0c9d800bf"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}