{"cells":[{"metadata":{"_uuid":"e6e40d451e28324c92a19ac50ba9049ba7627f93"},"cell_type":"markdown","source":"Tuning simple dense model to perform better....\n\n- batch normalization\n- drop out\n- selu\n- etc."},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nnp.set_printoptions(threshold=np.nan)\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\nprint(os.listdir(\"../input/embeddings\"))\nprint(os.listdir(\"../input/embeddings/GoogleNews-vectors-negative300\"))\n\n# Any results you write to the current directory are saved as output.\n\nimport gensim\nfrom gensim.utils import simple_preprocess\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.callbacks import Callback\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report,f1_score,precision_recall_fscore_support,recall_score,precision_score\nfrom keras import backend as K\nfrom sklearn.utils import class_weight\nimport matplotlib.pyplot as plt\n\n#https://www.kaggle.com/shujian/single-rnn-with-4-folds-v1-9\ndef threshold_search(y_true, y_proba):\n    best_threshold = 0\n    best_score = 0\n    for threshold in [i * 0.01 for i in range(100)]:\n        score = f1_score(y_true=y_true, y_pred=y_proba > threshold)\n        if score > best_score:\n            best_threshold = threshold\n            best_score = score\n    search_result = {'threshold': best_threshold, 'f1': best_score}\n    return search_result\n\n#inspired by https://www.tensorflow.org/tutorials/keras/basic_regression\nclass PrintDot(Callback):\n  def on_epoch_end(self, epoch, logs):\n    if epoch % 80 == 0: \n        print('')\n    print('.', end='')\n\n#inspiration from https://stackoverflow.com/questions/29760935/how-to-get-vector-for-a-sentence-from-the-word2vec-of-tokens-in-sentence\ndef load_x_from_df(df,model,max_len):\n    sequences = []\n    for question_text in df['question_text'].values:\n        tokens = simple_preprocess(question_text)\n        sentence = []\n        for word in tokens:\n            # print(model.wv[word])\n            if word in model.wv.vocab:\n                sentence.append(model.wv[word])\n        if len(sentence) == 0:\n            sentence = np.zeros((max_len,300))\n        sequences.append(np.mean(sentence,axis=1))\n    \n    return pad_sequences(sequences,dtype='float32',maxlen=max_len)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"print('loading word2vec model...')\nmodel = gensim.models.KeyedVectors.load_word2vec_format('../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin', binary=True)\nprint('vocab:',len(model.wv.vocab))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1a49eb61d312f9217b100cde6979e6191ddabe05"},"cell_type":"code","source":"df = pd.read_csv('../input/train.csv')\ndf[\"question_text\"].fillna(\"_##_\",inplace=True)\nmax_len = df['question_text'].apply(lambda x:len(x)).max()\nprint('max length of sequences:',max_len)\n# df = df.sample(frac=0.1)\n\nprint('columns:',df.columns)\npd.set_option('display.max_columns',None)\nprint('df head:',df.head())\nprint('example of the question text values:',df['question_text'].head().values)\nprint('what values contains target:',df.target.unique())\n\nprint('loading sequences...')\nprint('creating %d sequences'%len(df))\nx = load_x_from_df(df,model,max_len)\nprint(x.shape)\ny = df.target.values\nprint(y.shape)\nindexes_to_remove = []\nfor i in range(len(x)):\n    if np.sum(x[i]) == 0.:\n        indexes_to_remove.append(i)\n\nprint(indexes_to_remove)\n\n#when sequence contains only 0 masking would mask it and actually no input would be present for NN. So we need to remove those...\nif len(indexes_to_remove) > 0:\n    x = np.delete(x,indexes_to_remove,axis=0)\n    print(x.shape)\n    y = np.delete(y,indexes_to_remove)\n    print(y.shape)\n    \nx_train,x_test,y_train,y_test = train_test_split(x,y)\n\nprint(np.unique(y_train,return_counts=True))\nprint(np.unique(y_test,return_counts=True))\n\nprint('Computing class weights....')\n#https://datascience.stackexchange.com/questions/13490/how-to-set-class-weights-for-imbalanced-classes-in-keras\nclass_weights = class_weight.compute_class_weight('balanced',\n                                                 np.unique(y_test),\n                                                 y_test)\nprint('class_weights:',class_weights)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1f6c9ca76670d20aefc4f6567926712f37534e8b"},"cell_type":"code","source":"print('Creating model...')\n#inpiration from : https://github.com/keras-team/keras/blob/master/examples/imdb_fasttext.py\nfrom keras.models import Sequential\nfrom keras.layers import Dense\nfrom keras.layers import Masking\nfrom keras.layers import GlobalAveragePooling1D\nfrom keras.callbacks import EarlyStopping\n#4x256\n# {'threshold': 0.11, 'f1': 0.20848428957249193}\n#            1       0.16      0.31      0.21     20264\n#4x512\n# {'threshold': 0.13, 'f1': 0.20549967703238906}\n#               precision    recall  f1-score   support\n\n#            1       0.15      0.33      0.21     20264\n\n#back to relu\n#256 relu+batchnorm\n# {'threshold': 0.1, 'f1': 0.21432501711110813}\n#               precision    recall  f1-score   support\n#             1       0.15      0.40      0.21     20195\n# 32\n# {'threshold': 0.09, 'f1': 0.2132783413837488}\n#               precision    recall  f1-score   support\n#            1       0.15      0.38      0.21     20195\n\n#64\n# {'threshold': 0.09, 'f1': 0.21638974447963216}\n#               precision    recall  f1-score   support\n#            1       0.15      0.37      0.22     20195\n\nfrom keras.layers import BatchNormalization,Dropout,AlphaDropout\nfrom keras.engine import Layer\nfrom keras.initializers import Ones, Zeros\n\nclass LayerNormalization(Layer):\n    def __init__(self, **kwargs):\n        super(LayerNormalization, self).__init__(**kwargs)\n\n    def build(self, input_shape):\n        self.gain = self.add_weight(name='gain', shape=input_shape[-1:],\n                                    initializer=Ones(), trainable=True)\n        self.bias = self.add_weight(name='bias', shape=input_shape[-1:],\n                                    initializer=Zeros(), trainable=True)\n        super(LayerNormalization, self).build(input_shape)\n\n    def call(self, x, **kwargs):\n        mean = K.mean(x, axis=-1, keepdims=True)\n        std = K.std(x, axis=-1, keepdims=True)\n        # dot = *\n        # std+eps because of possible nans..\n        return self.gain * (x - mean) / (std + K.epsilon()) + self.bias\n\n    def compute_output_shape(self, input_shape):\n        return input_shape\n\n\ndnn_model = Sequential()\ndnn_model.add(BatchNormalization(input_shape=(x.shape[1],)))\ndnn_model.add(Dense(32, activation='relu'))\ndnn_model.add(LayerNormalization())\ndnn_model.add(Dense(1, activation='sigmoid'))\n\ndnn_model.compile(loss='binary_crossentropy',optimizer='adam',metrics=['accuracy'])\nprint(dnn_model.summary())\nprint('fiting model...')\nhistory = dnn_model.fit(x_train,y_train,\n                        validation_split=0.2,\n                        class_weight=class_weights,\n                        epochs=1000, \n                        callbacks=[EarlyStopping(patience=5)],\n                        verbose=2,\n                        batch_size = 64)\n\nprint('model score:',dnn_model.evaluate(x_test,y_test,verbose=0))\n\ny_pred = dnn_model.predict(x_test)\nsearch_result = threshold_search(y_test, y_pred)\nprint(search_result)\ny_pred = y_pred>search_result['threshold']\ny_pred = y_pred.astype(int)\n\nprint(classification_report(y_test,y_pred))\n\nprint(history.history.keys())\n\n_,ax = plt.subplots(1,2,figsize=(12,6))\nax[0].plot(history.history['loss'],label='loss')\nax[0].plot(history.history['val_loss'],label='val_loss')\nax[0].legend()\nax[0].set_title('loss')\nax[1].plot(history.history['acc'],label='acc')\nax[1].plot(history.history['val_acc'],label='val_acc')\nax[1].legend()\nax[1].set_title('acc')\nplt.show()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e518357d647f39e1ef5796133565c3856eaebdbe"},"cell_type":"code","source":"dnn_model.fit(x,y,class_weight=class_weights,epochs=50, verbose=2,batch_size = 64)\n\nprint('Loading test data...')\ndf_test = pd.read_csv('../input/test.csv')\ndf_test[\"question_text\"].fillna(\"_##_\",inplace=True)\n\nprint('creating %d sequences'%len(df_test))\nx_test = load_x_from_df(df_test,model,max_len)\nprint(x_test.shape)\n\nindexes_to_remove = []\nfor i in range(len(x_test)):\n    if np.sum(x_test[i]) == 0.:\n        indexes_to_remove.append(i)\n\nprint(indexes_to_remove)\n\n#when sequence contains only 0 masking would mask it and actually no input would be present for NN. So we need to remove those...\nif len(indexes_to_remove) > 0:\n    x_test = np.delete(x_test,indexes_to_remove,axis=0)\n    print(x_test.shape)\n\ny_pred = dnn_model.predict(x_test)\nprint(y_pred[:5] > search_result['threshold'])\ny_pred = y_pred > search_result['threshold']\ny_pred = y_pred.astype(int)\n\ndf_subm = pd.DataFrame()\ndf_subm['qid'] = df_test.qid\ndf_subm['prediction']=y_pred\nprint(df_subm.head())\ndf_subm.to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}