{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom json import JSONDecoder, JSONDecodeError  # for reading the JSON data files\nimport re  # for regular expressions\nimport os  # for os related operations","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def decode_obj(line, pos=0, decoder=JSONDecoder()):\n    no_white_space_regex = re.compile(r'[^\\s]')\n    while True:\n        match = no_white_space_regex.search(line, pos)\n        if not match:\n            return\n        pos = match.start()\n        try:\n            obj, pos = decoder.raw_decode(line, pos)\n        except JSONDecodeError as err:\n            print('Oops! something went wrong. Error: {}'.format(err))\n        yield obj","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_obj_with_last_n_val(line, n):\n    obj = next(decode_obj(line))  # type:dict\n    id = obj['id']\n    class_label = obj['classNum']\n\n    data = pd.DataFrame.from_dict(obj['values'])  # type:pd.DataFrame\n    data.set_index(data.index.astype(int), inplace=True)\n    last_n_indices = np.arange(0, 60)[-n:]\n    data = data.loc[last_n_indices]\n\n    return {'id': id, 'classType': class_label, 'values': data}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def convert_json_data_to_csv(data_dir: str, file_name: str):\n    \"\"\"\n    Generates a dataframe by concatenating the last values of each\n    multi-variate time series. This method is designed as an example\n    to show how a json object can be converted into a csv file.\n    :param data_dir: the path to the data directory.\n    :param file_name: name of the file to be read, with the extension.\n    :return: the generated dataframe.\n    \"\"\"\n    fname = os.path.join(data_dir, file_name)\n\n    all_df, labels, ids = [], [], []\n    with open(fname, 'r') as infile: # Open the file for reading\n        for line in infile:  # Each 'line' is one MVTS with its single label (0 or 1).\n            obj = get_obj_with_last_n_val(line, 1)\n            all_df.append(obj['values'])\n            labels.append(obj['classType'])\n            ids.append(obj['id'])\n\n    df = pd.concat(all_df).reset_index(drop=True)\n    df = df.assign(LABEL=pd.Series(labels))\n    df = df.assign(ID=pd.Series(ids))\n    df.set_index([pd.Index(ids)])\n    # Uncomment if you want to save this as CSV\n    # df.to_csv(file_name + '_last_vals.csv', index=False)\n    return df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"path_to_data = \"../input\"\nfile_name = \"fold1Training.json\"\n\ndf = convert_json_data_to_csv(path_to_data, file_name)  # shape: 27006 X 27\nprint('df.shape = {}'.format(df.shape))\n# print(list(df))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"path_to_data = \"../input\"\nfile_name = \"fold2Training.json\"\n\ndf1 = convert_json_data_to_csv(path_to_data, file_name)  # shape: 27006 X 27\nprint('df1.shape = {}'.format(df.shape))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"path_to_data = \"../input\"\nfile_name = \"testSet.json\"\n\ntest = convert_json_data_to_csv(path_to_data, file_name)  # shape: 27006 X 27\nprint('test.shape = {}'.format(df.shape))\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"df = df.dropna().append(df1.dropna())  # shape: 26666 X 27\nprint('df.shape = {}'.format(df.shape))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"t = np.int( np.round( (4/5) * df.shape[0],0 ) )\ndf_train = df[:t]  # shape: 18004 X 27\ndf_val = df[t:]  # shape: 9002 X 27\nprint('df_train.shape = {}'.format(df_train.shape))\nprint('df_val.shape = {}'.format(df_val.shape))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn import svm\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import f1_score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Separate values and labels columns\ndf_train_data = df_train.iloc[:, :-2]  # all columns excluding 'ID' and 'LABEL'\ndf_train_labels = pd.DataFrame(df_train.LABEL)  # only 'LABEL' column\n\ndf_val_data = df_val.iloc[:, :-2]  # all columns excluding 'ID' and 'LABEL'\ndf_val_labels = pd.DataFrame(df_val.LABEL)  # only 'LABEL' column\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nfrom sklearn.linear_model import LogisticRegression\n\nimport matplotlib.pyplot as plt\nfrom matplotlib.colors import ListedColormap\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.datasets import make_moons, make_circles, make_classification\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.gaussian_process import GaussianProcessClassifier\nfrom sklearn.gaussian_process.kernels import RBF\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier, AdaBoostClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.discriminant_analysis import QuadraticDiscriminantAnalysis\nfrom sklearn.svm import LinearSVC\nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.linear_model import Perceptron\nfrom sklearn.linear_model import PassiveAggressiveClassifier\nfrom sklearn.neighbors import NearestCentroid\nfrom sklearn.linear_model import RidgeClassifier\nfrom sklearn.naive_bayes import BernoulliNB, MultinomialNB\nfrom sklearn.metrics import f1_score,confusion_matrix\nimport xgboost as xgb\nfrom xgboost import XGBClassifier\nfrom sklearn.neural_network import MLPClassifier\n\nnames = [\"LR\",#\"MLP\",\n        #\"SVC\",        #\"SVC3\",\n        \"XGB\",\n         \"Passive-Aggressive\",    \n        \"linearSVC\",\"NearestCentroid\",\n        #\"multNB\",\n         \"bernouilliNB\",\n         #\"Ridge Classifier\",\n         \"Perceptron\",#\"kNN\",\n\n         \"SGD modeL2\",\"SGD elast\",\n         #\"Nearest Neighbors\",# \"Linear SVM\", \n         #\"RBF SVM\", #\"Gaussian Process\",\n         \"Decision Tree\", #\"Random Forest\", \n         #\"Neural Net\",\n        \"AdaBoost\",\n         #\"Naive Bayes\" #, \"QDA\"\n        ]\n\nclassifiers = [\n    LogisticRegression(),\n    #MLPClassifier(),\n    #SVC(kernel='linear'),\n    #SVC(kernel='sigmoid'),\n    XGBClassifier(learning_rate=0.1,n_estimators=100),\n    PassiveAggressiveClassifier(max_iter=50, tol=1e-3),    \n    LinearSVC(penalty=\"l2\", dual=False,tol=1e-3),\n    NearestCentroid(),\n    #MultinomialNB(alpha=.01),\n    BernoulliNB(alpha=.01),\n    #RidgeClassifier(tol=1e-2, solver=\"sag\"),\n    Perceptron(max_iter=50, tol=1e-3),\n    #KNeighborsClassifier(n_neighbors=10),\n\n    SGDClassifier(alpha=.0001, max_iter=50,penalty=\"l2\"),\n    SGDClassifier(alpha=.0001, max_iter=50,penalty=\"elasticnet\"),\n    #KNeighborsClassifier(5),\n    \n    #SVC(kernel=\"linear\", C=0.025),\n    #SVC(gamma=2, C=1),\n    #GaussianProcessClassifier(1.0 * RBF(1.0)),\n    DecisionTreeClassifier(max_depth=5),\n    #RandomForestClassifier(max_depth=5, n_estimators=10, max_features=1),\n    #MLPClassifier(alpha=1, max_iter=1000),\n    AdaBoostClassifier(),\n    #GaussianNB(),\n    #QuadraticDiscriminantAnalysis()\n    ]\n\n#or countmatrix or tfidfmatrix\n#X_train, X_test, y_train, y_test = train_test_split(countmatrix, y, test_size=0.2, random_state=42)\n    # iterate over classifiers\nfor name, clf in zip(names, classifiers):\n    clf.fit(df_train_data, np.ravel(df_train_labels))\n    score = clf.score(df_val_data,df_val_labels)\n    y_pred=clf.predict(df_val_data)\n    print(name,score,f1_score(df_val_labels,y_pred))\n    print('Confusion matrix:', confusion_matrix(df_val_labels,y_pred)  ) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Test the model against the validation set\npred_labels = clf.predict(df_val_data)\n\n# Evaluate the predictions\nscores = confusion_matrix(df_val_labels, pred_labels).ravel()\ntn, fp, fn, tp = scores\nprint('TN:{}\\tFP:{}\\tFN:{}\\tTP:{}'.format(tn, fp, fn, tp))\nf1 = f1_score(df_val_labels, pred_labels, average='binary', labels=[0, 1])\nprint('f1-score = {}'.format(f1))","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}