{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nimport os \n\nCLIP_PATH = \"/kaggle/input/seizure-prediction/Patient_1/Patient_1/\"\ndef get_clips(data_folder):# Get all clips\n    clips = os.listdir(data_folder)# Preictal recordings - time-series segments of the measurement before a seizure occura\n    clips_preictal = glob.glob(os.path.join(data_folder, \"*preictal*\"))# Interictial segments - segments with no oncoming seizures\n    clips_interictial = glob.glob(os.path.join(data_folder, \"*interictal*\"))\n    return clips_interictial, clips_preictal# Get EEG recordings\nclips_interictal, clips_preictal = get_clips(data_folder = CLIP_PATH)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-12T02:46:09.537216Z","iopub.execute_input":"2023-05-12T02:46:09.537889Z","iopub.status.idle":"2023-05-12T02:46:09.665419Z","shell.execute_reply.started":"2023-05-12T02:46:09.537857Z","shell.execute_reply":"2023-05-12T02:46:09.664094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-05-09T19:26:16.011433Z","iopub.execute_input":"2023-05-09T19:26:16.012162Z","iopub.status.idle":"2023-05-09T19:26:56.006668Z","shell.execute_reply.started":"2023-05-09T19:26:16.012129Z","shell.execute_reply":"2023-05-09T19:26:56.005672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nimport os \nimport numpy as np\nimport scipy.io\nfrom scipy.signal import stft\nfrom sklearn.model_selection import train_test_split\n\nCLIP_PATH = \"/kaggle/input/seizure-prediction/Patient_1/Patient_1/\"\n\ndef get_clips(data_folder):\n    # Get all clips\n    clips_preictal = sorted(glob.glob(os.path.join(data_folder, \"*preictal*\")))\n    clips_interictial = sorted(glob.glob(os.path.join(data_folder, \"*interictal*\")))\n    return clips_interictial, clips_preictal\n\n\ndef load_data():\n    \"\"\"\n    Load the EEG data from the .mat files and compute STFT for every second.\n    \"\"\"\n    types = ['Patient_1_interictal_segment', 'Patient_1_preictal_segment']\n    all_X = []\n    all_Y = []\n    for i, typ in enumerate(types):\n        for j in range(18):\n            fl = '/kaggle/input/seizure-prediction/Patient_1/Patient_1/{}_{}.mat'.format(typ, str(j + 1).zfill(4))\n            data = scipy.io.loadmat(fl)\n            k = typ.replace('Patient_1_', '') + '_'\n            d_array = data[k + str(j + 1)][0][0][0]\n            lst = list(range(3000000))  # 10 minutes\n            for m in lst[::5000]:\n                # Compute STFT every 1 second\n                p_secs = d_array[0][m:m+5000]\n                f, t, Sxx = stft(p_secs, fs=5000, window='hamming', nperseg=2048, noverlap=1024, return_onesided=False)\n                Sxx = np.log1p(np.abs(Sxx))\n                all_X.append(Sxx)\n                all_Y.append(i)\n\n    X = np.array(all_X)\n    y = np.array(all_Y)\n\n    return X, y\n\nX, y = load_data()\nprint(X.shape)\nprint(y.shape)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-12T02:46:11.768023Z","iopub.execute_input":"2023-05-12T02:46:11.768685Z","iopub.status.idle":"2023-05-12T02:47:04.333378Z","shell.execute_reply.started":"2023-05-12T02:46:11.768650Z","shell.execute_reply":"2023-05-12T02:47:04.332264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.utils import shuffle\n\nX, y = shuffle(X, y, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-12T02:47:13.147160Z","iopub.execute_input":"2023-05-12T02:47:13.147738Z","iopub.status.idle":"2023-05-12T02:47:13.745673Z","shell.execute_reply.started":"2023-05-12T02:47:13.147703Z","shell.execute_reply":"2023-05-12T02:47:13.744608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Splitting data into train/test, leaving only 600 samples for testing\nx_train = np.array(X[:21000])\ny_train = np.array(y[:21000])\nx_test = np.array(X[21000:])\ny_test = np.array(y[21000:])\n","metadata":{"execution":{"iopub.status.busy":"2023-05-12T02:47:15.761922Z","iopub.execute_input":"2023-05-12T02:47:15.762493Z","iopub.status.idle":"2023-05-12T02:47:16.553320Z","shell.execute_reply.started":"2023-05-12T02:47:15.762458Z","shell.execute_reply":"2023-05-12T02:47:16.552342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(x_train.shape)\nprint(y_train.shape)\nprint(x_test.shape)\nprint(y_test.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-12T02:47:18.987808Z","iopub.execute_input":"2023-05-12T02:47:18.988196Z","iopub.status.idle":"2023-05-12T02:47:18.993733Z","shell.execute_reply.started":"2023-05-12T02:47:18.988169Z","shell.execute_reply":"2023-05-12T02:47:18.992826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_values = np.unique(y_train)\nprint(unique_values)","metadata":{"execution":{"iopub.status.busy":"2023-05-12T02:47:21.060518Z","iopub.execute_input":"2023-05-12T02:47:21.061156Z","iopub.status.idle":"2023-05-12T02:47:21.067978Z","shell.execute_reply.started":"2023-05-12T02:47:21.061123Z","shell.execute_reply":"2023-05-12T02:47:21.066815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_values = np.unique(y_test)\nprint(unique_values)","metadata":{"execution":{"iopub.status.busy":"2023-05-12T02:47:22.833872Z","iopub.execute_input":"2023-05-12T02:47:22.834526Z","iopub.status.idle":"2023-05-12T02:47:22.840256Z","shell.execute_reply.started":"2023-05-12T02:47:22.834491Z","shell.execute_reply":"2023-05-12T02:47:22.839147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute the maximum value on the training set only\nmax_val = np.max(x_train)\n\n# Normalize both the training and test sets using the maximum value\nx_train = x_train / max_val\nx_test = x_test / max_val","metadata":{"execution":{"iopub.status.busy":"2023-05-12T02:47:24.864857Z","iopub.execute_input":"2023-05-12T02:47:24.865580Z","iopub.status.idle":"2023-05-12T02:47:27.312960Z","shell.execute_reply.started":"2023-05-12T02:47:24.865542Z","shell.execute_reply":"2023-05-12T02:47:27.311971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(x_train[:1])\nprint(x_test[:1])","metadata":{"execution":{"iopub.status.busy":"2023-05-12T02:47:29.005646Z","iopub.execute_input":"2023-05-12T02:47:29.006387Z","iopub.status.idle":"2023-05-12T02:47:29.012269Z","shell.execute_reply.started":"2023-05-12T02:47:29.006354Z","shell.execute_reply":"2023-05-12T02:47:29.011327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from __future__ import print_function\nimport tensorflow as tf\n\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, Flatten\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D\n#from keras import backend as K\n\nimport random\nimport numpy as np\nimport pandas as pd\n\nimport scipy.io\nfrom scipy.signal import spectrogram\nimport matplotlib.pyplot as plt\n","metadata":{"execution":{"iopub.status.busy":"2023-05-12T02:47:31.621108Z","iopub.execute_input":"2023-05-12T02:47:31.621774Z","iopub.status.idle":"2023-05-12T02:47:39.527218Z","shell.execute_reply.started":"2023-05-12T02:47:31.621739Z","shell.execute_reply":"2023-05-12T02:47:39.526145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Dropout, Flatten, Dense\n\ninput_shape = (2048, 6, 1) # Shape of input data\n\nmodel = Sequential()\n\n# Add convolutional layers\nmodel.add(Conv2D(32, kernel_size=(3, 3), activation='relu', input_shape=input_shape))\nmodel.add(Conv2D(32, kernel_size=(3, 3), activation='relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\n\n# Flatten output and add dense layers\nmodel.add(Flatten())\nmodel.add(Dense(32, activation='relu'))\nmodel.add(Dropout(0.5))\n\n# Output layer\nmodel.add(Dense(1, activation='sigmoid'))\n\n# Compile model\nmodel.compile(loss=tf.keras.losses.binary_crossentropy,\n              optimizer=tf.keras.optimizers.RMSprop(),\n              metrics=['accuracy'])\n","metadata":{"execution":{"iopub.status.busy":"2023-05-12T02:47:46.095219Z","iopub.execute_input":"2023-05-12T02:47:46.096744Z","iopub.status.idle":"2023-05-12T02:47:49.633364Z","shell.execute_reply.started":"2023-05-12T02:47:46.096699Z","shell.execute_reply":"2023-05-12T02:47:49.632449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-05-12T02:48:01.953608Z","iopub.execute_input":"2023-05-12T02:48:01.953966Z","iopub.status.idle":"2023-05-12T02:48:01.983737Z","shell.execute_reply.started":"2023-05-12T02:48:01.953939Z","shell.execute_reply":"2023-05-12T02:48:01.983014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 128\nnum_classes = 2\nepochs = 17\n\nmodel.fit(x_train, y_train,\n          batch_size=batch_size,\n          epochs=epochs,\n          verbose=1,\n          validation_data=(x_test, y_test))\n\nscore = model.evaluate(x_test, y_test, verbose=0)\nprint('Test loss:', score[0])\nprint('Test accuracy:', score[1])","metadata":{"execution":{"iopub.status.busy":"2023-05-12T02:48:04.445179Z","iopub.execute_input":"2023-05-12T02:48:04.445563Z","iopub.status.idle":"2023-05-12T02:50:30.065603Z","shell.execute_reply.started":"2023-05-12T02:48:04.445535Z","shell.execute_reply":"2023-05-12T02:50:30.064562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict the probabilities for the test data\ny_pred_proba = model.predict(x_test)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-12T02:50:34.920899Z","iopub.execute_input":"2023-05-12T02:50:34.921258Z","iopub.status.idle":"2023-05-12T02:50:35.229893Z","shell.execute_reply.started":"2023-05-12T02:50:34.921229Z","shell.execute_reply":"2023-05-12T02:50:35.228962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_pred_proba[:10])\nprint(y_test)","metadata":{"execution":{"iopub.status.busy":"2023-05-12T02:50:36.971389Z","iopub.execute_input":"2023-05-12T02:50:36.972121Z","iopub.status.idle":"2023-05-12T02:50:36.979599Z","shell.execute_reply.started":"2023-05-12T02:50:36.972084Z","shell.execute_reply":"2023-05-12T02:50:36.978633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc\n\nfpr, tpr, thresholds = roc_curve(y_test, y_pred_proba)\nroc_auc = auc(fpr, tpr)\n\n# Plot ROC curve\nplt.figure(figsize=(8,6))\nplt.plot(fpr, tpr, color='darkorange', lw=2, label='ROC curve (AUC = %0.2f)' % roc_auc)\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver operating characteristic (ROC) curve')\nplt.legend(loc=\"lower right\")\nplt.show()\n\n# Print AUC score\nprint(\"AUC score:\", roc_auc)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-12T02:50:42.494610Z","iopub.execute_input":"2023-05-12T02:50:42.495272Z","iopub.status.idle":"2023-05-12T02:50:43.066699Z","shell.execute_reply.started":"2023-05-12T02:50:42.495238Z","shell.execute_reply":"2023-05-12T02:50:43.065658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\nprint(os.getcwd())","metadata":{"execution":{"iopub.status.busy":"2023-05-12T02:56:29.970433Z","iopub.execute_input":"2023-05-12T02:56:29.971137Z","iopub.status.idle":"2023-05-12T02:56:29.979793Z","shell.execute_reply.started":"2023-05-12T02:56:29.971101Z","shell.execute_reply":"2023-05-12T02:56:29.976721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import joblib\n\njoblib.dump(model, 'model.pkl')","metadata":{"execution":{"iopub.status.busy":"2023-05-12T03:00:05.535797Z","iopub.execute_input":"2023-05-12T03:00:05.536192Z","iopub.status.idle":"2023-05-12T03:00:05.661235Z","shell.execute_reply.started":"2023-05-12T03:00:05.536155Z","shell.execute_reply":"2023-05-12T03:00:05.660125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install joblib","metadata":{"execution":{"iopub.status.busy":"2023-05-12T02:59:49.864208Z","iopub.execute_input":"2023-05-12T02:59:49.864958Z","iopub.status.idle":"2023-05-12T03:00:03.217223Z","shell.execute_reply.started":"2023-05-12T02:59:49.864923Z","shell.execute_reply":"2023-05-12T03:00:03.216010Z"},"trusted":true},"execution_count":null,"outputs":[]}]}