{"cells":[{"metadata":{"_uuid":"489563a49d3f21555c6dba6b265b98f10a69c6f0"},"cell_type":"markdown","source":"This is my first public kernel here. Thanks to other kagglers for showing the way how to do all this. Special thanks to\n\nhttps://www.kaggle.com/christofhenkel/how-to-preprocessing-when-using-embeddings for data preparation\n\nhttps://www.kaggle.com/guglielmocamporese/macro-f1-score-keras for f1 metric\n\nhttps://stackoverflow.com/questions/42918446/how-to-add-an-attention-mechanism-in-keras/44387553 for the attention layer\n"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"num_words=50000\nmax_len=64\n\n# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport re\nimport operator\nimport tensorflow as tf\nimport keras.backend as K\nfrom keras.engine.topology import Layer\nfrom keras import initializers, regularizers, constraints, optimizers, layers\nfrom tensorflow.keras.models import Sequential, Model\nfrom tensorflow.keras.layers import Input, concatenate, Dense, Flatten, Embedding, GRU, CuDNNGRU, LSTM, CuDNNLSTM, SpatialDropout1D, Dropout, Bidirectional, Conv1D, Activation, GlobalMaxPooling1D, GlobalAveragePooling1D, MaxPooling1D, RepeatVector, Permute\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nfrom tensorflow.python.keras.optimizers import Adam\nfrom tensorflow.keras import utils\nfrom keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom tqdm import tqdm\ntqdm.pandas()\nimport matplotlib.pyplot as plt\nimport tracemalloc\n%matplotlib inline \n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"24470aed7d17e852c1949b8f7bbbb142fdc00331"},"cell_type":"markdown","source":"**Loading and Preprocessing text**"},{"metadata":{"trusted":true,"_uuid":"1ed621249db3042c4f1867705266b5e48c5ac119"},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")\nprint(\"Train shape : \",train.shape)\nprint(\"Test shape : \",test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"530b8c56234a3ca379268ec38e0527eafc24eac9"},"cell_type":"code","source":"def build_vocab(sentences, verbose =  True):\n    \"\"\"\n    :param sentences: list of list of words\n    :return: dictionary of words and their count\n    \"\"\"\n    vocab = {}\n    for sentence in tqdm(sentences, disable = (not verbose)):\n        for word in sentence:\n            try:\n                vocab[word] += 1\n            except KeyError:\n                vocab[word] = 1\n    return vocab\ndef clean_text(x):\n\n    x = str(x)\n    for punct in \"/-'\":\n        x = x.replace(punct, ' ')\n    for punct in '&':\n        x = x.replace(punct, f' {punct} ')\n    for punct in '?!.,\"#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~' + '“”’':\n        x = x.replace(punct, '')\n    \n    return x\ndef clean_numbers(x):\n\n    x = re.sub('[0-9]{5,}', ' huge number ', x)\n    x = re.sub('[0-9]{4}', ' year ', x)\n    x = re.sub('[0-9]{3}', ' number ', x)\n    x = re.sub('[0-9]{2}', ' number ', x)\n    return x\n\ndef clean_more(x):\n    x=re.sub('\\s+', ' ', x).strip()\n    regex = re.compile('[^a-zA-Z] ')\n    #First parameter is the replacement, second parameter is your input string\n    return regex.sub('', x)\n    \n\n\n\n\ndef _get_mispell(mispell_dict):\n    mispell_re = re.compile('(%s)' % '|'.join(mispell_dict.keys()))\n    return mispell_dict, mispell_re\n\n\nmispell_dict = {'colour':'color',\n                'centre':'center',\n                'didnt':'did not',\n                'doesnt':'does not',\n                'isnt':'is not',\n                'shouldnt':'should not',\n                'favourite':'favorite',\n                'travelling':'traveling',\n                'counselling':'counseling',\n                'theatre':'theater',\n                'cancelled':'canceled',\n                'labour':'labor',\n                'organisation':'organization',\n                'citicise':'criticize',\n                }\nmispellings, mispellings_re = _get_mispell(mispell_dict)\n\ndef replace_typical_misspell(text):\n    def replace(match):\n        return mispellings[match.group(0)]\n\n    return mispellings_re.sub(replace, text)\n\ntrain[\"question_text\"] = train[\"question_text\"].progress_apply(lambda x: clean_text(x))\ntrain[\"question_text\"] = train[\"question_text\"].progress_apply(lambda x: clean_numbers(x))\ntrain[\"question_text\"] = train[\"question_text\"].progress_apply(lambda x: clean_more(x))\ntrain[\"question_text\"] = train[\"question_text\"].progress_apply(lambda x: x.lower())\ntrain[\"question_text\"] = train[\"question_text\"].progress_apply(lambda x: replace_typical_misspell(x))\nsentences = train[\"question_text\"].apply(lambda x: x.split())\n#to_remove = ['a','to','of','and']\n#sentences = [[word.lower() for word in sentence if not word.lower() in to_remove] for sentence in tqdm(sentences)]\n\n\ntest[\"question_text\"] = test[\"question_text\"].progress_apply(lambda x: clean_text(x))\ntest[\"question_text\"] = test[\"question_text\"].progress_apply(lambda x: clean_numbers(x))\ntest[\"question_text\"] = test[\"question_text\"].progress_apply(lambda x: clean_more(x))\ntest[\"question_text\"] = test[\"question_text\"].progress_apply(lambda x: x.lower())\ntest[\"question_text\"] = test[\"question_text\"].progress_apply(lambda x: replace_typical_misspell(x))\ntsentences = test[\"question_text\"].apply(lambda x: x.split())\n#to_remove = ['a','to','of','and']\n#tsentences = [[word.lower() for word in sentence if not word.lower() in to_remove] for sentence in tqdm(tsentences)]\n\nvocab = build_vocab(list(sentences)+list(tsentences))\ndef wordindex(vocab,n):\n  word_index={}\n  sorted_v = sorted(vocab.items(), key=operator.itemgetter(1))[::-1]\n  for i in range(n-3):\n    word_index[sorted_v[i][0]]=i+3\n  return(word_index)\n\nword_index=wordindex(vocab,num_words)\n\ndef zif(word):\n  ans=2\n  if (word in word_index):\n    ans=word_index[word]\n  return ans\n\nx_train = [[zif(word) for word in sentence] for sentence in tqdm(sentences)]\nx_train=pad_sequences(x_train, maxlen=max_len)\n\n\n\nx_test = [[zif(word) for word in sentence] for sentence in tqdm(tsentences)]\nx_test=pad_sequences(x_test, maxlen=max_len)\nprint({k: vocab[k] for k in list(vocab)[:5]})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8a03f88467c47df862fdad0b7d47cc511513d6a6"},"cell_type":"code","source":"def wordindex(vocab,n):\n  word_index={}\n  sorted_v = sorted(vocab.items(), key=operator.itemgetter(1))[::-1]\n  for i in range(n-3):\n    word_index[sorted_v[i][0]]=i+3\n  return(word_index)\n\nword_index=wordindex(vocab,num_words)\n\ndef zif(word):\n  ans=2\n  if (word in word_index):\n    ans=word_index[word]\n  return ans\n\nx_train = [[zif(word) for word in sentence] for sentence in tqdm(sentences)]\nx_train=pad_sequences(x_train, maxlen=max_len)\n\n\n\nx_test = [[zif(word) for word in sentence] for sentence in tqdm(tsentences)]\nx_test=pad_sequences(x_test, maxlen=max_len)\n\ny_train=train['target']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c4e23b62accff390448f262927b3e88d8c0c54d4"},"cell_type":"code","source":"del train\ndel sentences\ndel tsentences","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":false},"cell_type":"markdown","source":"**Embeddings**"},{"metadata":{"trusted":true,"_uuid":"bd8ce0b3d149e138532a0e8243af840afb52b73f"},"cell_type":"code","source":"embedding_matrix_Glove = np.zeros((num_words, 300))\nwith open('../input/embeddings/glove.840B.300d/glove.840B.300d.txt', 'r') as f:\n    for line in tqdm(f):\n        values = line.split()\n        word = values[0]\n        if (word in word_index):\n            try:\n                word_vector = np.asarray(values[1:], dtype='float32')        \n            except ValueError:\n                pass  # do nothing!\n            else:\n                embedding_matrix_Glove[word_index[word]] = word_vector","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f8e040bec94931bfbd763210a04a816a79b6163c"},"cell_type":"code","source":"embedding_matrix_Wiki = np.zeros((num_words, 300))\nwith open('../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec', 'r') as f:\n    for line in tqdm(f):\n        values = line.split()\n        word = values[0]\n        if (word in word_index):\n            try:\n                word_vector = np.asarray(values[1:], dtype='float32')        \n            except ValueError:\n                pass  # do nothing!\n            else:\n                embedding_matrix_Wiki[word_index[word]] = word_vector","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7710ee20c809b18f9d54db805eed8984a8cc4129"},"cell_type":"markdown","source":"**NET**"},{"metadata":{"trusted":true,"_uuid":"912cf1751dc2db9e985aab0920054fceabdc3dfd"},"cell_type":"code","source":"len(y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"861bf1d3ef32188ec7585955119ff63041d6c02b"},"cell_type":"code","source":"def f1(y_true, y_pred):\n    y_pred = K.round(y_pred+0.15)\n    tp = K.sum(K.cast(y_true*y_pred, 'float'), axis=0)\n    tn = K.sum(K.cast((1-y_true)*(1-y_pred), 'float'), axis=0)\n    fp = K.sum(K.cast((1-y_true)*y_pred, 'float'), axis=0)\n    fn = K.sum(K.cast(y_true*(1-y_pred), 'float'), axis=0)\n\n    p = tp / (tp + fp + K.epsilon())\n    r = tp / (tp + fn + K.epsilon())\n\n    f1 = 2*p*r / (p+r+K.epsilon())\n    f1 = tf.where(tf.is_nan(f1), tf.zeros_like(f1), f1)\n    return K.mean(f1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"62fe9422d85a56c969b2bd906b872822a83031ef"},"cell_type":"code","source":"class Netn:  \n    def __init__(self,em):\n        tweet_input = Input(shape=(max_len,), dtype='int32')\n        tweet_encoder = Embedding(num_words, 300, input_length=max_len,\n                          weights=[em], trainable=False)(tweet_input)\n        X = SpatialDropout1D(0.1)(tweet_encoder)\n        X = Bidirectional(CuDNNGRU(64, return_sequences=True))(X)\n        activations = CuDNNGRU(64, return_sequences=True)(X)\n        # compute importance for each step\n        attention = Dense(1, activation='tanh')(activations)\n        attention = Flatten()(attention)\n        attention = Activation('softmax')(attention)\n        attention = RepeatVector(64)(attention)\n        attention = Permute([2, 1])(attention)\n        attention = Dropout(0.5)(attention)\n        X = concatenate([activations, attention])\n        x=CuDNNGRU(64, return_sequences=True)(X)\n        a=GlobalMaxPooling1D()(x)\n        b=GlobalAveragePooling1D()(x) \n        x=concatenate([a,b])\n        x=Dense(64, activation='relu')(x)\n        x=Dropout(0.4)(x)\n        output = Dense(1, activation='sigmoid')(x)\n        self.model = Model(inputs=[tweet_input], outputs=[output])\n        self.model.summary()\n    \n    def unfreeze(self):\n        self.model.layers[1].trainable = True\n  \n    def fit(self,b1,b2,epp,bs,**data):\n        self.model.compile(**data)\n        \n        x_t=x_train[b1:b2]\n        y_t=y_train[b1:b2]\n        if (epp>0):\n            filepath='tmp.hd5'\n            cp=ModelCheckpoint(filepath, monitor=\"val_f1\",verbose=1, save_best_only=True,mode='max')\n            history=self.model.fit(x_t, \n                    y_t, \n                    epochs=epp,\n                    batch_size=bs,\n                    callbacks=[cp],\n                    validation_split=0.1)\n            self.model.load_weights(filepath)\n            plt.plot(history.history['f1'], label='f1 train')\n            plt.plot(history.history['val_f1'], label='f1 val')\n            plt.xlabel('epoche')\n            plt.ylabel('f1')\n            plt.legend()\n            plt.show()\n            ansz=history.history['val_f1']\n            mx=max(ansz)\n            print('val f1 is maximal {} on a step {}'.format(mx,ansz.index(mx)+1))\n\n  \n\n    def predvec(self,x_test):\n        return(self.model.predict(x_test))    \n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3de3fe2f9ff9188e1624741aa78c18e8318e2bfc"},"cell_type":"code","source":"m1=Netn(embedding_matrix_Wiki)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"803657c9e6f1e3f881031d35cf9f5deb70487429"},"cell_type":"code","source":"m1.fit(0,600000,3,2000,loss='binary_crossentropy', metrics=['accuracy', f1],\n              optimizer=Adam(lr=1e-3))\nm1.fit(0,1250000,5,2000,loss='binary_crossentropy', metrics=['accuracy', f1],\n              optimizer=Adam(lr=3e-4))\nm1.unfreeze()\nm1.fit(0,1250000,14,2000,loss='binary_crossentropy', metrics=['accuracy', f1],\n              optimizer=Adam(lr=3e-5))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5405ecd0de6d310382d34edba99b6daac0e200b1"},"cell_type":"code","source":"m2=Netn(embedding_matrix_Glove)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c79449dd2f01d25069adbf6746cfc61c29f3e3c5"},"cell_type":"code","source":"m2.fit(0,600000,3,2000,loss='binary_crossentropy', metrics=['accuracy', f1],\n              optimizer=Adam(lr=1e-3))\nm2.fit(0,1250000,5,2000,loss='binary_crossentropy', metrics=['accuracy', f1],\n              optimizer=Adam(lr=3e-4))\nm2.unfreeze()\nm2.fit(0,1250000,14,2000,loss='binary_crossentropy', metrics=['accuracy', f1],\n              optimizer=Adam(lr=3e-5))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6ce71cd475a285905d5e7490028e2cc6329fc8e1"},"cell_type":"markdown","source":"**Best threshold**"},{"metadata":{"trusted":true,"_uuid":"06e6e40ad1c1889ba7f016633d384ec18b7037b7"},"cell_type":"code","source":"lval=1250000\ny_v=y_train.tolist()[lval:]\nx_v=x_train[lval:]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c8b24402c3d3b96fbe1a831a3ea763f55496b374"},"cell_type":"code","source":"u1=m1.predvec(x_v)\nu2=m2.predvec(x_v)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4d9e2c17ec38f2f50074c3dac34608927e2d39d1"},"cell_type":"code","source":"def qf1(t,vv):\n    tp=0\n    fp=0\n    fn=0\n    yt=y_train.tolist()\n    for i in range(len(vv)):\n        if vv[-i]>t:\n            if yt[-i]==1:\n                tp+=1\n            else:\n                fp+=1\n        else:\n            if yt[-i]==1:\n                fn+=1\n    return (2*tp/(2*tp+fn+fp))\n\ndef best(vv):\n    cc=0.35\n    a=0\n    for i in range(10):\n        r= qf1(cc+0.01*i,vv)\n        if (r>a):\n            ii=i\n            a=r\n    a=0\n    for i in range(10):\n        r= qf1(cc+0.01*(ii-1)+0.002*i,vv)\n        if (r>a):\n            iii=i\n            a=r\n    \n    print(a)\n    return(cc+0.01*(ii-1)+0.002*iii)\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"322cc4dc78595c5c5392e066d3d0f0a45d5066c0"},"cell_type":"code","source":"bt=best((u1+u2)/2)\nprint(bt)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f58837f1ed7abced7a424e94da6c050a69ba6ee1"},"cell_type":"markdown","source":"**Committung**"},{"metadata":{"trusted":true,"_uuid":"75e55fb30af9fc749462f876627a29522d9c6ae2"},"cell_type":"code","source":"def ans(v,tr):\n    res=np.zeros(len(v))\n    for i in range(len(v)):\n        if v[i]>tr:\n            res[i]=1\n    return res","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b323fa699214bae001e1c1cb49397833414cfcd5"},"cell_type":"code","source":"v1=m1.predvec(x_test)\nv2=m2.predvec(x_test)\nvv=ans((v1+v2)/2,bt)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d8be27fb2a48461560c1eafb6f4e8cd0c2028502"},"cell_type":"code","source":"out = np.column_stack((test['qid'].values,vv))\nnp.savetxt('submission.csv', out, header=\"qid,prediction\", \n            comments=\"\", fmt=\"%s,%d\")","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}