{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7654739,"sourceType":"datasetVersion","datasetId":4462774},{"sourceId":236702525,"sourceType":"kernelVersion"},{"sourceId":236702533,"sourceType":"kernelVersion"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd, numpy as np\nfrom glob import glob\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-29T19:34:23.771148Z","iopub.execute_input":"2025-04-29T19:34:23.772330Z","iopub.status.idle":"2025-04-29T19:34:23.776839Z","shell.execute_reply.started":"2025-04-29T19:34:23.772300Z","shell.execute_reply":"2025-04-29T19:34:23.775885Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#code for GRDA DATA for 1 Participant Table\nBASE_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/'\n\ndf = pd.DataFrame({'path': glob(BASE_PATH + '**/*.parquet')})\ndf['test_type'] = df['path'].str.split('/').str.get(-2).str.split('_').str.get(-1)\ndf['id'] = df['path'].str.split('/').str.get(-1).str.split('.').str.get(0)\n\ndf_GRDA = pd.read_parquet(BASE_PATH + 'train_eegs/4030851372.parquet')\ndf_GRDA.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T19:34:23.784445Z","iopub.execute_input":"2025-04-29T19:34:23.785195Z","iopub.status.idle":"2025-04-29T19:34:24.276778Z","shell.execute_reply.started":"2025-04-29T19:34:23.785158Z","shell.execute_reply":"2025-04-29T19:34:24.275728Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#code for GRDA Fp1 for 1 participant Spectrogram\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom scipy.signal import spectrogram\nimport numpy as np\n\nchannel_name = 'Fp1' \nsignal = df_GRDA[channel_name].dropna().values \n\nfs = 200  \n\nfrequencies, times, Sxx = spectrogram(signal, fs=fs, nperseg=256, noverlap=128)\n\nplt.figure(figsize=(10, 4))\nplt.pcolormesh(times, frequencies, 10 * np.log10(Sxx + 1e-10), shading='gouraud')\nplt.title(f'GRDA Spectrogram of {channel_name}')\nplt.ylabel('Frequency [Hz]')\nplt.xlabel('Time [sec]')\nplt.colorbar(label='Intensity [dB]')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T19:34:24.278852Z","iopub.execute_input":"2025-04-29T19:34:24.279220Z","iopub.status.idle":"2025-04-29T19:34:24.939092Z","shell.execute_reply.started":"2025-04-29T19:34:24.279194Z","shell.execute_reply":"2025-04-29T19:34:24.938162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Code for Seizure Data for 1 participant Table\nBASE_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/'\n\ndf = pd.DataFrame({'path': glob(BASE_PATH + '**/*.parquet')})\ndf['test_type'] = df['path'].str.split('/').str.get(-2).str.split('_').str.get(-1)\ndf['id'] = df['path'].str.split('/').str.get(-1).str.split('.').str.get(0)\n\ndf_seizure = pd.read_parquet(BASE_PATH + 'train_eegs/266631836.parquet')\ndf_seizure.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T19:34:24.939899Z","iopub.execute_input":"2025-04-29T19:34:24.940154Z","iopub.status.idle":"2025-04-29T19:34:25.289138Z","shell.execute_reply.started":"2025-04-29T19:34:24.940134Z","shell.execute_reply":"2025-04-29T19:34:25.288310Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#code for Spectrogram of Seizure Fp1 Data of 1 Participant\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom scipy.signal import spectrogram\nimport numpy as np\n\n\nsignal = df_seizure[channel_name].dropna().values  \n\nfs = 200  \n\nfrequencies, times, Sxx = spectrogram(signal, fs=fs, nperseg=256, noverlap=128)\n\nplt.figure(figsize=(10, 4))\nplt.pcolormesh(times, frequencies, 10 * np.log10(Sxx + 1e-10), shading='gouraud')\nplt.title(f'Seizure Spectrogram of {channel_name}')\nplt.ylabel('Frequency [Hz]')\nplt.xlabel('Time [sec]')\nplt.colorbar(label='Intensity [dB]')\nplt.tight_layout()      \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T19:34:25.290055Z","iopub.execute_input":"2025-04-29T19:34:25.290379Z","iopub.status.idle":"2025-04-29T19:34:25.897423Z","shell.execute_reply.started":"2025-04-29T19:34:25.290353Z","shell.execute_reply":"2025-04-29T19:34:25.896376Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Code for LPD Data for 1 participant Table\nBASE_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/'\n\ndf = pd.DataFrame({'path': glob(BASE_PATH + '**/*.parquet')})\ndf['test_type'] = df['path'].str.split('/').str.get(-2).str.split('_').str.get(-1)\ndf['id'] = df['path'].str.split('/').str.get(-1).str.split('.').str.get(0)\n\ndf_LPD = pd.read_parquet(BASE_PATH + 'train_eegs/1317431280.parquet')\ndf_LPD.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T19:34:25.899776Z","iopub.execute_input":"2025-04-29T19:34:25.900182Z","iopub.status.idle":"2025-04-29T19:34:26.440627Z","shell.execute_reply.started":"2025-04-29T19:34:25.900148Z","shell.execute_reply":"2025-04-29T19:34:26.439759Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Code for Spectrogram of LPD Fp1 Data for 1 Participant \nimport numpy as np\n\nchannel_name = 'Fp1' \nsignal = df_LPD[channel_name].dropna().values  \n\nfs = 200  \n\nfrequencies, times, Sxx = spectrogram(signal, fs=fs, nperseg=256, noverlap=128)\n\nplt.figure(figsize=(10, 4))\nplt.pcolormesh(times, frequencies, 10 * np.log10(Sxx + 1e-10), shading='gouraud')\nplt.title(f'LPD Spectrogram of {channel_name}')\nplt.ylabel('Frequency [Hz]')\nplt.xlabel('Time [sec]')\nplt.colorbar(label='Intensity [dB]')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T19:34:26.441888Z","iopub.execute_input":"2025-04-29T19:34:26.442230Z","iopub.status.idle":"2025-04-29T19:34:27.373049Z","shell.execute_reply.started":"2025-04-29T19:34:26.442200Z","shell.execute_reply":"2025-04-29T19:34:27.372163Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Code for GPD Data for 1 participant Table\nBASE_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/'\n\ndf = pd.DataFrame({'path': glob(BASE_PATH + '**/*.parquet')})\ndf['test_type'] = df['path'].str.split('/').str.get(-2).str.split('_').str.get(-1)\ndf['id'] = df['path'].str.split('/').str.get(-1).str.split('.').str.get(0)\n\ndf_GPD = pd.read_parquet(BASE_PATH + 'train_eegs/2846570074.parquet')\ndf_GPD.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T20:10:03.060367Z","iopub.execute_input":"2025-04-29T20:10:03.060871Z","iopub.status.idle":"2025-04-29T20:10:03.471389Z","shell.execute_reply.started":"2025-04-29T20:10:03.060846Z","shell.execute_reply":"2025-04-29T20:10:03.470353Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Code for Spectrogram of GPD Fp1 Data for 1 Participant \nimport numpy as np\n\nchannel_name = 'Fp1'\nsignal = df_GPD[channel_name].dropna().values \n\nfs = 200  \n\nfrequencies, times, Sxx = spectrogram(signal, fs=fs, nperseg=256, noverlap=128)\n\nplt.figure(figsize=(10, 4))\nplt.pcolormesh(times, frequencies, 10 * np.log10(Sxx + 1e-10), shading='gouraud')\nplt.title(f'GPD Spectrogram of {channel_name}')\nplt.ylabel('Frequency [Hz]')\nplt.xlabel('Time [sec]')\nplt.colorbar(label='Intensity [dB]')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T20:10:09.969850Z","iopub.execute_input":"2025-04-29T20:10:09.970904Z","iopub.status.idle":"2025-04-29T20:10:10.598459Z","shell.execute_reply.started":"2025-04-29T20:10:09.970873Z","shell.execute_reply":"2025-04-29T20:10:10.597160Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Code for LRDA Data for 1 participant Table\nBASE_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/'\n\ndf = pd.DataFrame({'path': glob(BASE_PATH + '**/*.parquet')})\ndf['test_type'] = df['path'].str.split('/').str.get(-2).str.split('_').str.get(-1)\ndf['id'] = df['path'].str.split('/').str.get(-1).str.split('.').str.get(0)\n\ndf_LRDA = pd.read_parquet(BASE_PATH + 'train_eegs/2222924277.parquet')\ndf_LRDA.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T19:34:28.473585Z","iopub.execute_input":"2025-04-29T19:34:28.473896Z","iopub.status.idle":"2025-04-29T19:34:28.826346Z","shell.execute_reply.started":"2025-04-29T19:34:28.473869Z","shell.execute_reply":"2025-04-29T19:34:28.825628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Code for Spectrogram of LRDA Fp1 Data for 1 Participant \nimport numpy as np\n\nchannel_name = 'Fp1'\nsignal = df_LRDA[channel_name].dropna().values \n\nfs = 200  \n\nfrequencies, times, Sxx = spectrogram(signal, fs=fs, nperseg=256, noverlap=128)\n\nplt.figure(figsize=(10, 4))\nplt.pcolormesh(times, frequencies, 10 * np.log10(Sxx + 1e-10), shading='gouraud')\nplt.title(f'LRDA Spectrogram of {channel_name}')\nplt.ylabel('Frequency [Hz]')\nplt.xlabel('Time [sec]')\nplt.colorbar(label='Intensity [dB]')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T19:34:28.827040Z","iopub.execute_input":"2025-04-29T19:34:28.827292Z","iopub.status.idle":"2025-04-29T19:34:29.583629Z","shell.execute_reply.started":"2025-04-29T19:34:28.827273Z","shell.execute_reply":"2025-04-29T19:34:29.582737Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Code for LRDA and Seizure Fp1 Data Table\nimport numpy as np\nimport pandas as pd\n\neeg_data = {\n    '266631836': {'path': '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/266631836.parquet', 'label': 'Seizure'},\n    '2222924277': {'path': '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/2222924277.parquet', 'label': 'LRDA'},\n    '2894007647': {'path': '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/2894007647.parquet', 'label': 'Seizure'},\n    '722738444': {'path': '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/722738444.parquet', 'label': 'LRDA'},\n    '338161210': {'path': '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/338161210.parquet', 'label': 'LRDA'},\n    '2088807520': {'path': '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/2088807520.parquet', 'label': 'Seizure'},\n    '3030710864': {'path': '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/3030710864.parquet', 'label': 'Seizure'},\n    '3190279138': {'path': '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/3190279138.parquet', 'label': 'Seizure'},\n    '1844014178': {'path': '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1844014178.parquet', 'label': 'LRDA'},\n    '2622179549': {'path': '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/2622179549.parquet', 'label': 'LRDA'},\n}\n\nmin_length = None\nfor pid, info in eeg_data.items():\n    try:\n        df = pd.read_parquet(info['path'])\n        if 'Fp1' in df.columns:\n            length = len(df['Fp1'])\n            if min_length is None or length < min_length:\n                min_length = length\n    except Exception as e:\n        print(f\"Failed to load {pid}: {e}\")\n\nprint(f\"Minimum Fp1 length: {min_length}\")\ndata_rows = []\n\nfor pid, info in eeg_data.items():\n    try:\n        df = pd.read_parquet(info['path'])\n        if 'Fp1' not in df.columns:\n            print(f\"Skipping {pid} — no Fp1 column\")\n            continue\n\n        fp1_values = df['Fp1'].values[:min_length]\n        row = {\n            'Patient ID': pid,\n            'Label': info['label']\n        }\n        for t in range(min_length):\n            row[f'Time {t}'] = fp1_values[t]\n\n        data_rows.append(row)\n    except Exception as e:\n        print(f\"Skipping {pid} due to error: {e}\")\n\nstructured_df = pd.DataFrame(data_rows)\n\nprint(f\"Final table shape: {structured_df.shape}\")\nstructured_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T19:34:29.584623Z","iopub.execute_input":"2025-04-29T19:34:29.584920Z","iopub.status.idle":"2025-04-29T19:34:29.931835Z","shell.execute_reply.started":"2025-04-29T19:34:29.584892Z","shell.execute_reply":"2025-04-29T19:34:29.931087Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Code for PCA Table of Seizure and LRDA Fp1 Data\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\n\nfeatures_only = structured_df.drop(columns=['Patient ID', 'Label']).fillna(structured_df.mean(numeric_only=True))\n\nscaler = StandardScaler()\nscaled_features = scaler.fit_transform(features_only)\n\npca = PCA(n_components=2)\nprincipal_components = pca.fit_transform(scaled_features)\n\npca_df = pd.DataFrame(principal_components, columns=['PC1', 'PC2'])\npca_df['Patient ID'] = structured_df['Patient ID'].values\npca_df = pca_df.set_index('Patient ID')\npca_df['Label'] = structured_df['Label'].values\n\npca_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T19:34:29.932628Z","iopub.execute_input":"2025-04-29T19:34:29.932842Z","iopub.status.idle":"2025-04-29T19:34:33.815041Z","shell.execute_reply.started":"2025-04-29T19:34:29.932826Z","shell.execute_reply":"2025-04-29T19:34:33.814142Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Code for Spectrogram of PCA Fp1 Seizure v. LRDA data\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nsns.set(style='whitegrid')\n\nplt.figure(figsize=(10, 7))\nsns.scatterplot(\n    data=pca_df,\n    x='PC1',\n    y='PC2',\n    hue='Label',\n    palette={'Seizure': 'red', 'LRDA': 'blue'},\n    s=100,\n    edgecolor='black'\n)\n\nplt.title('PCA of EEG Data (Fp1) - Seizure vs LRDA', fontsize=16)\nplt.xlabel('Principal Component 1', fontsize=12)\nplt.ylabel('Principal Component 2', fontsize=12)\nplt.legend(title='Label')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T19:34:33.816268Z","iopub.execute_input":"2025-04-29T19:34:33.816582Z","iopub.status.idle":"2025-04-29T19:34:34.191683Z","shell.execute_reply.started":"2025-04-29T19:34:33.816559Z","shell.execute_reply":"2025-04-29T19:34:34.190810Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Code for spectrogram that Classifies Fake Points into the PCA FP1 Data\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.preprocessing import LabelEncoder\n\nle = LabelEncoder()\ny_encoded = le.fit_transform(pca_df['Label']) \n\nX = pca_df[['PC1', 'PC2']].values\nX = np.array(X).astype(float)\n\nclf = DecisionTreeClassifier(random_state=42)\nclf.fit(X, y_encoded)\n\nx_min, x_max = X[:, 0].min() - 1, X[:, 0].max() + 1\ny_min, y_max = X[:, 1].min() - 1, X[:, 1].max() + 1\nxx, yy = np.meshgrid(np.linspace(x_min - 35, x_max, 300),\n                     np.linspace(y_min - 30, y_max, 300))\n\nZ = clf.predict(np.c_[xx.ravel(), yy.ravel()])\nZ = Z.reshape(xx.shape)\n\nplt.figure(figsize=(10, 7))\nplt.contourf(xx, yy, Z, alpha=0.3, cmap='coolwarm')\n\nsns.set_style(\"white\")\nsns.scatterplot(\n    x=X[:, 0], y=X[:, 1], hue=pca_df['Label'],\n    palette={'Seizure': 'red', 'LRDA': 'blue'},\n    edgecolor='black', s=100\n)\n\nn_random_points = 20\nrandom_points = np.random.uniform(\n    low=(x_min, y_min),\n    high=(x_max, y_max),\n    size=(n_random_points, 2)\n)\n\nrandom_preds = clf.predict(random_points)\npredicted_labels = le.inverse_transform(random_preds)\ncolor_map = {'Seizure': 'red', 'LRDA': 'blue'}\n\nfor i, (point, label) in enumerate(zip(random_points, predicted_labels)):\n    plt.scatter(point[0], point[1], \n                color=color_map[label], \n                s=150, \n                edgecolor='black', \n                marker='X',\n                label='Random Test Point' if i == 0 else \"\"\n               )\n\nplt.title(\"Decision Boundary: PCA Components (Seizure vs LRDA) Fp1\", fontsize=16)\nplt.xlabel(\"PC1\", fontsize=12)\nplt.ylabel(\"PC2\", fontsize=12)\nplt.legend()\nplt.grid(False)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T19:34:34.194158Z","iopub.execute_input":"2025-04-29T19:34:34.194395Z","iopub.status.idle":"2025-04-29T19:34:34.690539Z","shell.execute_reply.started":"2025-04-29T19:34:34.194378Z","shell.execute_reply":"2025-04-29T19:34:34.689693Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#We then wanted to look at the T3 Data only for LRDA and Seizure Participants\n#The T3 data is in the other notebook","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T19:34:34.691821Z","iopub.execute_input":"2025-04-29T19:34:34.692136Z","iopub.status.idle":"2025-04-29T19:34:34.695815Z","shell.execute_reply.started":"2025-04-29T19:34:34.692108Z","shell.execute_reply":"2025-04-29T19:34:34.694964Z"}},"outputs":[],"execution_count":null}]}