{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":8011206,"sourceType":"datasetVersion","datasetId":4719156}],"dockerImageVersionId":30674,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2024-04-02T20:31:23.266204Z","iopub.execute_input":"2024-04-02T20:31:23.266544Z","iopub.status.idle":"2024-04-02T20:31:23.270798Z","shell.execute_reply.started":"2024-04-02T20:31:23.266517Z","shell.execute_reply":"2024-04-02T20:31:23.269822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def TrainPreProcessing(DataPortion, Validation_Split):\n    \n    # Load all numpy arrays \n    EEG_Sig_Total = np.load('/kaggle/input/logsig3scaled/eeg_data_scaled.npy')   \n    NumVotes_Total = np.load('/kaggle/input/logsig3scaled/num_votes.npy')\n    Targets_Total = np.load('/kaggle/input/logsig3scaled/targets.npy')\n\n    n_features = 2470\n    \n    # Combine columns into one array: ¦368 Features¦6 Targets¦NumVotes¦ \n    TotalDataset = np.hstack((EEG_Sig_Total, Targets_Total, NumVotes_Total))\n    \n    # We want to drop all rows with nans in them\n    nan_rows = np.isnan(TotalDataset).any(axis=1)\n    # Drop rows with NaN values\n    TotalDataset = TotalDataset[~nan_rows]\n    \n    TotalN = TotalDataset.shape[0]\n\n    # shuffle rows\n    np.random.seed(69)\n    p = np.random.permutation(TotalN)\n\n    TotalDataset = TotalDataset[p]\n\n    # Work with only a small portion of the dataset for experimentation \n    N = round(DataPortion*TotalN)\n\n    TruncatedDataset = TotalDataset[:N]\n    \n    # set proportion of train data for validation set\n    Validation_N = round(Validation_Split*N)\n\n    ValidationDataset = TruncatedDataset[:Validation_N]\n    TrainDataset = TruncatedDataset[Validation_N:]\n\n    X_val, y_val = ValidationDataset[:,:n_features], ValidationDataset[:,n_features:n_features+6]\n    X_train, y_train = TrainDataset[:,:n_features], TrainDataset[:,n_features:n_features+6]\n    N_votes_train = TrainDataset[:,-1]\n\n    return X_train, y_train, X_val, y_val, N_votes_train\n","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2024-04-02T20:31:23.404616Z","iopub.execute_input":"2024-04-02T20:31:23.404973Z","iopub.status.idle":"2024-04-02T20:31:23.414345Z","shell.execute_reply.started":"2024-04-02T20:31:23.404940Z","shell.execute_reply":"2024-04-02T20:31:23.413344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# adjust data portion accordingly (1 = 100% of the dataset)\n\nX_train, y_train, X_test, y_test, N_votes_train = TrainPreProcessing(DataPortion=1, Validation_Split=0.1)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T20:42:03.591701Z","iopub.execute_input":"2024-04-02T20:42:03.592057Z","iopub.status.idle":"2024-04-02T20:42:06.325978Z","shell.execute_reply.started":"2024-04-02T20:42:03.592027Z","shell.execute_reply":"2024-04-02T20:42:06.324920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport random\nfrom tensorflow.keras import layers, Model\nfrom tensorflow.keras.losses import kullback_leibler_divergence\nfrom tensorflow.keras.callbacks import EarlyStopping\n\n\n# Define the number of input features and output classes\nnum_features = 2470\nnum_classes = 6","metadata":{"execution":{"iopub.status.busy":"2024-04-02T20:42:07.819135Z","iopub.execute_input":"2024-04-02T20:42:07.819535Z","iopub.status.idle":"2024-04-02T20:42:07.826168Z","shell.execute_reply.started":"2024-04-02T20:42:07.819509Z","shell.execute_reply":"2024-04-02T20:42:07.825411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set random seeds for reproducibility\nseed_value = 69\n\n# 1. Set the seed for Python's built-in random number generator\nrandom.seed(seed_value)\n# 2. Set the seed for NumPy\nnp.random.seed(seed_value)\n# 3. Set the seed for TensorFlow\ntf.random.set_seed(seed_value)\n\n# Performance on total train dataset w/ 10% portioned for test: 0.4381\n\n\n\n# Random attempt\ndef create_model():\n    inputs = tf.keras.Input(shape=(num_features,))\n    \n    x = layers.Dense(800, activation='sigmoid')(inputs) # 800 is good\n    x = layers.BatchNormalization()(x)\n    x = layers.Dropout(0.5)(x)\n    \n    x = layers.Dense(500, activation='relu')(x) # 500 is good\n    x = layers.BatchNormalization()(x)\n    x = layers.Dropout(0.2)(x)  \n    \n    x = layers.Dense(500, activation='relu')(x) # 500 is good\n    x = layers.Dropout(0.2)(x)  \n    \n    x = layers.Dense(350, activation='relu')(x) # 350 is good   \n    x = layers.Dropout(0.1)(x)\n\n    outputs = layers.Dense(num_classes, activation='softmax')(x)\n    model = Model(inputs, outputs)\n    return model\n\n# Instantiate the model\nmodel = create_model()\n\n# Compile the model with KL divergence loss\nmodel.compile(optimizer='adam',\n              loss=kullback_leibler_divergence)\n\n# Print the model summary\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-04-02T20:45:44.089503Z","iopub.execute_input":"2024-04-02T20:45:44.090194Z","iopub.status.idle":"2024-04-02T20:45:44.224370Z","shell.execute_reply.started":"2024-04-02T20:45:44.090158Z","shell.execute_reply":"2024-04-02T20:45:44.223526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"early_stopping = EarlyStopping(monitor='val_loss', patience=20, restore_best_weights=True)\n\n# Train the model\nhistory = model.fit(X_train, y_train, batch_size=500, epochs=1000, validation_split=0.1, callbacks=[early_stopping])\n\n# Evaluate the model on test data\nloss = model.evaluate(X_test, y_test)\nprint(\"Test Loss:\", loss)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-04-02T20:45:44.251366Z","iopub.execute_input":"2024-04-02T20:45:44.251867Z","iopub.status.idle":"2024-04-02T20:47:29.181135Z","shell.execute_reply.started":"2024-04-02T20:45:44.251839Z","shell.execute_reply":"2024-04-02T20:47:29.180143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 0.3 w/ 0.1 test\n\n0.713","metadata":{"execution":{"iopub.status.busy":"2024-04-02T20:47:29.187343Z","iopub.execute_input":"2024-04-02T20:47:29.187671Z","iopub.status.idle":"2024-04-02T20:47:29.194184Z","shell.execute_reply.started":"2024-04-02T20:47:29.187643Z","shell.execute_reply":"2024-04-02T20:47:29.193029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Plot training & validation loss values\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title('Model loss')\nplt.ylabel('KLD Loss')\nplt.xlabel('Epoch')\nplt.legend(['Train', 'Validation'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-02T21:48:57.281175Z","iopub.execute_input":"2024-04-02T21:48:57.282020Z","iopub.status.idle":"2024-04-02T21:48:57.612887Z","shell.execute_reply.started":"2024-04-02T21:48:57.281975Z","shell.execute_reply":"2024-04-02T21:48:57.610918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(X_train)","metadata":{"execution":{"iopub.status.busy":"2024-04-01T22:55:03.796236Z","iopub.status.idle":"2024-04-01T22:55:03.796562Z","shell.execute_reply.started":"2024-04-01T22:55:03.796402Z","shell.execute_reply":"2024-04-01T22:55:03.796416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}