{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport warnings\nimport gc\n\nimport pyarrow.parquet as pq\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom tqdm import tqdm_notebook as tqdm\nfrom sklearn.model_selection import train_test_split\nfrom keras.models import Model\nfrom tqdm import tqdm # Processing time measurement\nfrom keras import backend as K # The backend give us access to tensorflow operations and allow us to create the Attention class\nfrom keras import optimizers # Allow us to access the Adam class to modify some parameters\nfrom sklearn.model_selection import GridSearchCV, StratifiedKFold # Used to use Kfold to train our model\nfrom keras.callbacks import * # This object helps the model to train in a smarter way, avoiding overfitting\nfrom keras.layers import * # Keras is the most friendly Neural Network library, this Kernel use a lot of layers classes\nfrom keras.models import Model\n\n%matplotlib inline\nwarnings.filterwarnings('ignore')\nprint(os.listdir(\"../input\"))\nprint(os.listdir(\"./\"))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = '../input/sswwaatt/swat_train.csv'\ntest_path = '../input/sswwaatt/swat_test.csv'\ntrain_data = pd.read_csv(train_path, header = None)\ntest_data = pd.read_csv(test_path, header = None)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 使用imlbearn库中上采样方法中的SMOTE接口\n# from imblearn.over_sampling import SMOTE\n# # 定义SMOTE模型，random_state相当于随机数种子的作用\n# smo = SMOTE(random_state=42)\n# X_smo, y_smo = smo.fit_sample(test_data.iloc[:,:-1],test_data.iloc[:,-1] )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from collections import Counter\n# print(Counter(y_smo))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def MinMaxScaler(data):\n    \"\"\"Min Max normalizer.\n    Args:\n      - data: original data\n    Returns:\n      - norm_data: normalized data\n    \"\"\"\n    numerator = data - np.min(data, 0)\n    denominator = np.max(data, 0) - np.min(data, 0)\n    norm_data = numerator / (denominator + 1e-7)\n    return norm_data","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def multivariate_data(dataset, target, start_index,\n                      end_index, history_size, step):\n    data = []\n    labels = []\n    if end_index is None:\n        end_index = len(dataset) - history_size + 1\n\n    for start, stop in zip(range(0, end_index), range(history_size, len(dataset) + 1)):\n        data.append(dataset[start:stop, :])\n\n    for i in range(start_index, end_index):\n        labels.append(np.isin(1, target[i:i + history_size]))\n\n    return np.array(data), np.array(labels, dtype=np.float)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seq_len = 32\ntrain_dataset = train_data.values\ntest_dataset = test_data.values","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, y_train = multivariate_data(train_dataset[:, :-1], train_dataset[:, -1], 0,\n                                         None, history_size=seq_len, step=1)\nx_test, y_test = multivariate_data(test_dataset[:, :-1], test_dataset[:, -1], 0,\n                                         None, history_size=seq_len, step=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"###  test ######\ntest = pd.read_excel('../input/testttttt/2123.xlsx', header = None).values\nprint(test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test, y_t = multivariate_data(test[:, :-1], test[:, -1], 0,\n                                         None, history_size=2, step=1)\nprint(x_test, y_t)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(x_train.shape, y_train.shape)\n# print(x_test.shape, y_test.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import Counter\nprint(Counter(y_test))\nprint(Counter(y_train))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from keras.utils import to_categorical\n# nb_class = 2\n# y_test = to_categorical(y_test, nb_class)\n# y_train = to_categorical(y_train, nb_class)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.models import Sequential\nfrom keras import layers\n\nfrom keras.optimizers import Adam\nfrom keras.callbacks import ModelCheckpoint, LearningRateScheduler, EarlyStopping, ReduceLROnPlateau\nimport keras.models as models\nimport keras.backend as K\nimport keras","metadata":{"_uuid":"434b85fcbddefc7e7852020909bf2a98bdd6c58a","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n\ndef mcc(y_true, y_pred):\n    cm = confusion_matrix(y_true, y_pred)\n    TP = cm[0][0]\n    FP = cm[0][1]\n    FN = cm[1][0]\n    TN = cm[1][1]\n    val = ((TP * TN) - (FP * FN)) / ((TP + FP)*(TP + FN)*(TN + FP)*(TN + FN))**0.5\n    return val","metadata":{"_uuid":"6279b47e4dc18482e4ebc26aefd115c21477473f","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def matthews_correlation(y_true, y_pred):\n    '''Calculates the Matthews correlation coefficient measure for quality\n    of binary classification problems.\n    '''\n    y_pred_pos = K.round(K.clip(y_pred, 0, 1))\n    y_pred_neg = 1 - y_pred_pos\n\n    y_pos = K.round(K.clip(y_true, 0, 1))\n    y_neg = 1 - y_pos\n\n    tp = K.sum(y_pos * y_pred_pos)\n    tn = K.sum(y_neg * y_pred_neg)\n\n    fp = K.sum(y_neg * y_pred_pos)\n    fn = K.sum(y_pos * y_pred_neg)\n\n    numerator = (tp * tn - fp * fn)\n    denominator = K.sqrt((tp + fp) * (tp + fn) * (tn + fp) * (tn + fn))\n\n    return numerator / (denominator + K.epsilon())","metadata":{"_uuid":"3b93dbf9fed506113f89e7a779677b6e3c7d30b3","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport pyarrow.parquet as pq\nimport os \nimport numpy as np\nfrom keras.layers import * \nfrom keras.models import Model\nfrom tqdm import tqdm \nfrom sklearn.model_selection import train_test_split \nfrom keras import backend as K \nfrom keras import optimizers \nfrom sklearn.model_selection import GridSearchCV, StratifiedKFold \nfrom keras.callbacks import * ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_model(input_shape):\n    input_layer = Input(shape=(input_shape[1], input_shape[2],))\n    \n    x = Conv1D(32, 8, padding='same',activation='relu')(input_layer)\n    x = MaxPooling1D(2, padding='same')(x)\n    x = Conv1D(64, 8, padding='same', activation='relu')(x)\n    x = MaxPooling1D(2, padding='same')(x)\n    x = Conv1D(128, 8, padding='same', activation='relu')(x)\n    x = MaxPooling1D(2, padding='same')(x)\n    x = Conv1D(256, 8, padding='same', activation='relu')(x)\n    \n    x = LSTM(64, dropout = 0.2, recurrent_dropout = 0.5)(x)\n#     x = Dense(32, activation=\"relu\")(x)\n\n    output_layer = Dense(1, activation=\"sigmoid\")(x)\n    model = Model(inputs=input_layer, outputs=output_layer)\n    model.compile(loss='binary_crossentropy',\n              optimizer='adam',\n              metrics=['accuracy',matthews_correlation])\n\n    return model","metadata":{"_uuid":"835bf0f11e59dee510dd036621785c2362384c81","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N_SPLITS = 5\nsplits = list(StratifiedKFold(n_splits=N_SPLITS, shuffle=True, random_state=2019).split(x_train, y_train))\npreds_val = []\ny_val = []\nfor idx, (train_idx, val_idx) in enumerate(splits):\n    K.clear_session() \n    print(\"Beginning fold {}\".format(idx+1))\n    train_X, train_y, val_X, val_y = x_train[train_idx], y_train[train_idx], x_train[val_idx], y_train[val_idx]\n    model = make_model(train_X.shape)\n    ckpt = ModelCheckpoint('weights_{}.h5'.format(idx), save_best_only=True, save_weights_only=True, verbose=1, monitor='val_matthews_correlation', mode='max')\n    model.fit(train_X, train_y, batch_size=32, epochs=10, validation_data=[val_X, val_y], callbacks=[ckpt])\n    model.load_weights('weights_{}.h5'.format(idx))\n    preds_val.append(model.predict(val_X, batch_size=32))\n    y_val.append(val_y)\n   \npreds_val = np.concatenate(preds_val)[...,0]\ny_val = np.concatenate(y_val)\npreds_val.shape, y_val.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # The output of this kernel must be binary (0 or 1), but the output of the NN Model is float (0 to 1).\n# # So, find the best threshold to convert float to binary is crucial to the result\n# # this piece of code is a function that evaluates all the possible thresholds from 0 to 1 by 0.01\n# def threshold_search(y_true, y_proba):\n#     best_threshold = 0\n#     best_score = 0\n#     for threshold in tqdm([i * 0.01 for i in range(100)]):\n#         score = K.eval(matthews_correlation(y_true.astype(np.float64), (y_proba > threshold).astype(np.float64)))\n#         if score > best_score:\n#             best_threshold = threshold\n#             best_score = score\n#     search_result = {'threshold': best_threshold, 'matthews_correlation': best_score}\n#     return search_result","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# best_threshold = threshold_search(y_val, preds_val)['threshold']","metadata":{"_uuid":"fba17e77694a35fe3fdddc631fced82357bd7d06","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# best_threshold","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preds_test = []\n# for i in range(N_SPLITS):\n#     model.load_weights('weights_{}.h5'.format(i))\n#     pred = model.predict(x_test, batch_size=300, verbose=1)\n#     pred_3 = []\n#     for pred_scalar in pred:\n#         for i in range(1):\n#             pred_3.append(pred_scalar)\n#     preds_test.append(pred_3)","metadata":{"_uuid":"f0bab392881faea748f0d615e184dd15266f2eec","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preds = (np.squeeze(np.mean(preds_test, axis=0)) > best_threshold).astype(np.int)\n# preds.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_val_pred = y_val_pred.flatten()\n# y_val_pred[y_val_pred >= 0.8] = 1\n# y_val_pred[y_val_pred < 0.8] = 0\n# y_val_pred.sum()","metadata":{"_uuid":"e8863147279006caffcd3bf0333c07ce7d3b106d","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_test","metadata":{"_uuid":"5bc5441bdf51378e4d9a98de3fb66c3fe01fda09","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# mcc(y_test,preds)","metadata":{"_uuid":"cdd3d11f6c8a6a2224d9c296286d695449e00054","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.metrics import confusion_matrix\n# tn, fp, fn, tp = confusion_matrix(y_test, preds).ravel()\n# precision = tp / (tp+fp)\n# print(precision)\n# recall = tp / (tp+fn)\n# print(recall)\n# accuracy = (tp+tn) / (tp+tn+fp+fn)\n# print(accuracy)\n# f1 = (2 * precision * recall) / (precision + recall)\n# print(f1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = np.load('../input/vsb-lstmattention/X.npy')\ny = np.load('../input/vsb-lstmattention/y.npy')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N_SPLITS = 5\nsplits = list(StratifiedKFold(n_splits=N_SPLITS, shuffle=True, random_state=2019).split(X, y))\npreds_val = []\ny_val = []\nfor idx, (train_idx, val_idx) in enumerate(splits):\n    K.clear_session() \n    print(\"Beginning fold {}\".format(idx+1))\n    train_X, train_y, val_X, val_y = X[train_idx], y[train_idx], X[val_idx], y[val_idx]\n    model = make_model(train_X.shape)\n    ckpt = ModelCheckpoint('weights_{}.h5'.format(idx), save_best_only=True, save_weights_only=True, verbose=1, monitor='val_matthews_correlation', mode='max')\n    model.fit(train_X, train_y, batch_size=32, epochs=20, validation_data=[val_X, val_y], callbacks=[ckpt])\n    model.load_weights('weights_{}.h5'.format(idx))\n    preds_val.append(model.predict(val_X, batch_size=32))\n    y_val.append(val_y)\n   \npreds_val = np.concatenate(preds_val)[...,0]\ny_val = np.concatenate(y_val)\npreds_val.shape, y_val.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Add predict label to test data and train again.","metadata":{"_uuid":"7e1f1407e65f0cbd5696f64e869d8c0023469c45","trusted":true}}]}