{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import nltk\nimport pandas as pd\nimport numpy as np\nimport operator \nimport re\nfrom tqdm import tqdm\ntqdm.pandas()\nfrom keras.preprocessing.text import Tokenizer","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7a7bbdd4e8c7f127ff72bbdb503eea8aceada533","scrolled":true},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")\ndf = pd.concat([train ,test])\n\nprint(\"Number of texts: \", df.shape[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4dd4b321337f22e5755181aa8aac53cff6601880"},"cell_type":"code","source":"def load_embed(file):\n    def get_coefs(word,*arr): \n        return word, np.asarray(arr, dtype='float32')\n    if file == '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec':\n        embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(file) if len(o)>100)\n    else:\n        embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(file, encoding='latin'))\n    return embeddings_index\n\nparagram =  '../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt'\nprint(\"Extracting Paragram embedding\")\nembed_paragram = load_embed(paragram)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"be73ccc623a80bb80b8bebcaab581c6104b9a63c"},"cell_type":"code","source":"df['lowered_question'] = df['question_text'].apply(lambda x: x.lower())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"303ae019813a57e478f2b5d2782045597d060350"},"cell_type":"code","source":"# build a contraction mapping dictionary \ncontraction_mapping = {\"ain't\": \"is not\", \"aren't\": \"are not\",\"can't\": \"cannot\", \"'cause\": \"because\", \"could've\": \"could have\", \"couldn't\": \"could not\", \"didn't\": \"did not\",  \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\", \"he'd\": \"he would\",\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\", \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\",  \"I'd\": \"I would\", \"I'd've\": \"I would have\", \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"I've\": \"I have\", \"i'd\": \"i would\", \"i'd've\": \"i would have\", \"i'll\": \"i will\",  \"i'll've\": \"i will have\",\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'd've\": \"it would have\", \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\", \"ma'am\": \"madam\", \"mayn't\": \"may not\", \"might've\": \"might have\",\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \"must've\": \"must have\", \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \"needn't\": \"need not\", \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\", \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\", \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\", \"she'll\": \"she will\", \"she'll've\": \"she will have\", \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\", \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \"there'd've\": \"there would have\", \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\", \"they'll\": \"they will\", \"they'll've\": \"they will have\", \"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\", \"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\", \"we'll've\": \"we will have\", \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\", \"what're\": \"what are\",  \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\", \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\"you'd\": \"you would\", \"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\" }\ndef known_contractions(embed):\n    known = []\n    for contract in contraction_mapping:\n        if contract in embed:\n            known.append(contract)\n    return known\n# clean contractions in text data before embedding \ndef clean_contractions(text, mapping):\n    specials = [\"’\", \"‘\", \"´\", \"`\"]\n    for s in specials:\n        text = text.replace(s, \"'\")\n    text = ' '.join([mapping[t] if t in mapping else t for t in text.split(\" \")])\n    return text\ndf['treated_question'] = df['lowered_question'].progress_apply(lambda x: clean_contractions(x, contraction_mapping))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"296c37291de339406ed54372835b096891f68375"},"cell_type":"code","source":"punct = \"/-'?!.,#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~\" + '\"\"“”’' + '∞θ÷α•à−β∅³π‘₹´°£€\\×™√²—–&'\npunct_mapping = {\"‘\": \"'\", \"₹\": \"e\", \"´\": \"'\", \"°\": \"\", \"€\": \"e\", \"™\": \"tm\", \"√\": \" sqrt \", \"×\": \"x\", \"²\": \"2\", \"—\": \"-\", \"–\": \"-\", \"’\": \"'\", \"_\": \"-\", \"`\": \"'\", '“': '\"', '”': '\"', '“': '\"', \"£\": \"e\", '∞': 'infinity', 'θ': 'theta', '÷': '/', 'α': 'alpha', '•': '.', 'à': 'a', '−': '-', 'β': 'beta', '∅': '', '³': '3', 'π': 'pi','\\u200b': ' ', '…': ' ... ', '\\ufeff': '', 'करना': '', 'है': '' }\ndef clean_special_chars(text, punct, mapping):\n    \n    ## use a map to replace unknown characters with known ones.\n    for p in mapping:\n        text = text.replace(p, mapping[p])\n    ## make sure there are spaces between words and punctuation\n    for p in punct:\n        text = text.replace(p, f' {p} ')\n        \n    return text\n\ndf['treated_question'] = df['treated_question'].progress_apply(lambda x: clean_special_chars(x, punct, punct_mapping))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"7e4bdd64bcf94b2e6bd1aea47d0a5927357aec6f"},"cell_type":"code","source":"# build a word mapping dictionary for frequent mispells \nmispell_dict = {'colour': 'color', 'centre': 'center', 'favourite': 'favorite', 'travelling': 'traveling', 'counselling': 'counseling', 'theatre': 'theater', 'cancelled': 'canceled', 'labour': 'labor', 'organisation': 'organization', 'wwii': 'world war 2', 'citicise': 'criticize', 'youtu ': 'youtube ', 'Qoura': 'Quora', 'sallary': 'salary', 'Whta': 'What', 'narcisist': 'narcissist', 'howdo': 'how do', 'whatare': 'what are', 'howcan': 'how can', 'howmuch': 'how much', 'howmany': 'how many', 'whydo': 'why do', 'doI': 'do I', 'theBest': 'the best', 'howdoes': 'how does', 'mastrubation': 'masturbation', 'mastrubate': 'masturbate', \"mastrubating\": 'masturbating', 'pennis': 'penis', 'Etherium': 'Ethereum', 'narcissit': 'narcissist', 'bigdata': 'big data', '2k17': '2017', '2k18': '2018', 'qouta': 'quota', 'exboyfriend': 'ex boyfriend', 'airhostess': 'air hostess', \"whst\": 'what', 'watsapp': 'whatsapp', 'demonitisation': 'demonetization', 'demonitization': 'demonetization', 'demonetisation': 'demonetization'}\ndef correct_spelling(x, dic):\n    for word in dic.keys():\n        x = x.replace(word, dic[word])\n    return x\n\ndf['treated_question'] = df['treated_question'].progress_apply(lambda x: correct_spelling(x, mispell_dict))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ccccb3a121bac798cf7f7e103f25ffefcdd9f063"},"cell_type":"code","source":"import re\ndef clean_numbers(x):\n    if re.search(\"[0-9]+\", x) != None:\n        x = re.sub('[0-9]+',' {} '.format(re.search('[0-9]+',x).group()),x) \n    x = re.sub('[0-9]{5,}', '#####', x)\n    x = re.sub('[0-9]{4}', '####', x)\n    x = re.sub('[0-9]{3}', '###', x)\n    x = re.sub('[0-9]{2}', '##', x)\n    return x\ndf[\"treated_question_num\"] = df[\"treated_question\"].progress_apply(lambda x: clean_numbers(x))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"41080f8d64ecb7d66cc546eb447cab56a7053646"},"cell_type":"code","source":"# modified from load_glove\ndef build_embedding_matrix_v2(word_index, embed):\n    all_embs = np.stack(embed.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n    \n    empty_vector = np.random.normal(emb_mean, emb_std,(embed_size,))\n    return np.vstack([empty_vector, all_embs])\nembedding_matrix = build_embedding_matrix_v2(_, embed_paragram)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0759b355663ca05e15e8931270dfc4eded36a53c"},"cell_type":"code","source":"word_index = {word:i for i,word in enumerate(embed_paragram.keys(),1)}\n\ndef label_sentence(s, word_index):\n    return [word_index.get(x,0) for x in s.split()]\n\ntrain_embed = df[:train.shape[0]][['qid', 'target', 'treated_question_num']]\ntest_embed = df[train.shape[0]:][['qid', 'treated_question_num']]\n\nfrom sklearn.model_selection import train_test_split\ntrain, val = train_test_split(train_embed, test_size=0.1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3fa91cd869e38429374416975aa83e36acc98a99"},"cell_type":"code","source":"maxlen = 60\nfrom keras.preprocessing.sequence import pad_sequences\ny_train = train['target']\ny_val = val['target']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1e123b6edd16b62e59bac972eb18e94b20fc1abb","scrolled":true},"cell_type":"code","source":"%%time\ntrain_X = train['treated_question_num'].apply(label_sentence, args=(word_index,))\ntrain_X = pad_sequences(train_X, maxlen=maxlen)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"034e0c342debbe44a126d03cc1fa7a0da4839dbd"},"cell_type":"code","source":"%%time\nval_X = val['treated_question_num'].apply(label_sentence, args=(word_index,))\nval_X = pad_sequences(val_X, maxlen=maxlen)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bc6251a642e83b38a005a9d37d003100bede6ba5"},"cell_type":"code","source":"%%time\nsub_data = test_embed['treated_question_num'].apply(label_sentence, args=(word_index,))\nsub_data = pad_sequences(sub_data, maxlen=maxlen)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"408e168b28668f2faf049f8ac98a4eff42518b71"},"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\nimport gc\n\ndel df, train_embed, test_embed\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"98b714d944c6449e9cd7bea5189dbdfe95a5ba90"},"cell_type":"code","source":"# del df, train_embed, test_embed","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"88080286091331cbc919074bb67b05cc8b89df0c"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"1cd473d8982389ba5b25c1fd9383f346cf7068ed"},"cell_type":"code","source":"# model = keras.Sequential()\n# model.add(keras.layers.Embedding(embedding_matrix.shape[0], 300, weights=[embedding_matrix], input_length=maxlen, trainable=False))\n# model.add(keras.layers.GlobalAveragePooling1D())\n# model.add(keras.layers.Dense(16, activation=tf.nn.relu))\n# model.add(keras.layers.Dense(1, activation=tf.nn.sigmoid))\n# model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"c50567b3f974a439c8ba0b5757616119f513338f"},"cell_type":"code","source":"# model.compile(optimizer=tf.train.AdamOptimizer(),\n#               loss='binary_crossentropy',\n#               metrics=['accuracy'])\n# early_stop = keras.callbacks.EarlyStopping(monitor='val_loss',\n#                               min_delta=0,\n#                               patience=1,\n#                               verbose=0, mode='auto')\n\n# history = model.fit(train_X,\n#                 y_train,\n#                 epochs=100,\n#                 batch_size=1024,\n#                 validation_data=(val_X, y_val),\n#                 verbose=1,\n#                 callbacks=[early_stop,])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ebae653e63d7f8fb7f9259482ec6726a3fc340ad"},"cell_type":"code","source":"# from sklearn import metrics\n# y_pred = model.predict(val_X, batch_size=1024, verbose=1)\n# best_thresh = 0\n# best_score = 0\n# for thresh in np.arange(0.1, 0.9, 0.01):\n#     thresh = np.round(thresh, 2)\n#     score = metrics.f1_score(y_val, (y_pred>thresh).astype(int))\n#     if score>best_score:\n#         best_score=score\n#         best_thresh=thresh\n#     print(\"F1 score at threshold {0} is {1}\".format(thresh, score))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"711ab029a621acd25c70be62ca70fc0d2c1b7b59"},"cell_type":"markdown","source":"# LSTM "},{"metadata":{"trusted":true,"_uuid":"4f76f27cb66c83428b635fffc8cf2dc17a44844a"},"cell_type":"code","source":"#del model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c370f8cd40655025a46b31551776cbba55374d12"},"cell_type":"code","source":"import gc\ngc.collect()\ntf.reset_default_graph()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4238562d963f02670771aa28ee70086b285c2ab4"},"cell_type":"code","source":"# tf.keras.backend.clear_session()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4d1a36b5558fa08f9b2d1e1d08ae5eb61d2d388e"},"cell_type":"code","source":"model = keras.Sequential()\nmodel.add(keras.layers.Embedding(embedding_matrix.shape[0], 300, weights=[embedding_matrix], input_length=maxlen, trainable=False))\n# model.add(keras.layers.Bidirectional(keras.layers.LSTM(64, return_sequences=True)))\nmodel.add(keras.layers.Bidirectional(keras.layers.LSTM(64)))\nmodel.add(keras.layers.Dense(1, activation=tf.nn.sigmoid))\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e0f84b94201fc426caf508aba78c64ea2d2ede02"},"cell_type":"code","source":"model.compile(optimizer=tf.train.AdamOptimizer(),\n              loss='binary_crossentropy',\n              metrics=['accuracy'])\nearly_stop = keras.callbacks.EarlyStopping(monitor='val_loss',\n                              min_delta=0,\n                              patience=1,\n                              verbose=0, mode='auto')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7c726fdc2e1b97cf16e456f1128ede55d3fe930f","scrolled":true},"cell_type":"code","source":"history = model.fit(train_X,\n                y_train,\n                epochs=100,\n                batch_size=1024,\n                validation_data=(val_X, y_val),\n                verbose=1,\n                callbacks=[early_stop,])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4bb1b85e556caf83f455f9a855276530d58f906c"},"cell_type":"code","source":"from sklearn import metrics\ny_pred = model.predict(val_X, batch_size=1024, verbose=1)\nbest_thresh = 0\nbest_score = 0\nfor thresh in np.arange(0.1, 0.9, 0.01):\n    thresh = np.round(thresh, 2)\n    score = metrics.f1_score(y_val, (y_pred>thresh).astype(int))\n    if score>best_score:\n        best_score=score\n        best_thresh=thresh\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, score))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2e0fd30ef6c7f61151bb34b66dd21abfb305dd59"},"cell_type":"code","source":"pred_test_y = model.predict([sub_data], batch_size=1024, verbose=0)\nsub = pd.read_csv('../input/sample_submission.csv')\nsub.prediction = pred_test_y > best_thresh\nsub.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1eb93e90588e9223eabdb67ec89aca5afa15e2cc"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}