{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7654739,"sourceType":"datasetVersion","datasetId":4462774},{"sourceId":236702525,"sourceType":"kernelVersion"},{"sourceId":236702533,"sourceType":"kernelVersion"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Code for LRDA Data for 1 participant Table\nimport os\nimport pandas as pd, numpy as np\nfrom glob import glob\nimport matplotlib.pyplot as plt\n\nBASE_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/'\n\ndf = pd.DataFrame({'path': glob(BASE_PATH + '**/*.parquet')})\ndf['test_type'] = df['path'].str.split('/').str.get(-2).str.split('_').str.get(-1)\ndf['id'] = df['path'].str.split('/').str.get(-1).str.split('.').str.get(0)\n\ndf_LRDA = pd.read_parquet(BASE_PATH + 'train_eegs/2222924277.parquet')\ndf_LRDA.head()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-02T05:50:51.229574Z","iopub.execute_input":"2025-05-02T05:50:51.229921Z","iopub.status.idle":"2025-05-02T05:50:51.586725Z","shell.execute_reply.started":"2025-05-02T05:50:51.229899Z","shell.execute_reply":"2025-05-02T05:50:51.585794Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Code for Spectrogram of LRDA T3 Data for 1 Participant\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom scipy.signal import spectrogram\n\nchannel_name = 'T3' \nsignal = df_LRDA[channel_name].dropna().values \n\nfs = 200  \n\nfrequencies, times, Sxx = spectrogram(signal, fs=fs, nperseg=256, noverlap=128)\n\nplt.figure(figsize=(10, 4))\nplt.pcolormesh(times, frequencies, 10 * np.log10(Sxx + 1e-10), shading='gouraud')\nplt.title(f'LRDA Spectrogram of {channel_name}')\nplt.ylabel('Frequency [Hz]')\nplt.xlabel('Time [sec]')\nplt.colorbar(label='Intensity [dB]')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T05:50:44.222960Z","iopub.execute_input":"2025-05-02T05:50:44.223176Z","iopub.status.idle":"2025-05-02T05:50:44.976215Z","shell.execute_reply.started":"2025-05-02T05:50:44.223161Z","shell.execute_reply":"2025-05-02T05:50:44.975328Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Code for Seizure Data for 1 participant Table\nBASE_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/'\n\ndf = pd.DataFrame({'path': glob(BASE_PATH + '**/*.parquet')})\ndf['test_type'] = df['path'].str.split('/').str.get(-2).str.split('_').str.get(-1)\ndf['id'] = df['path'].str.split('/').str.get(-1).str.split('.').str.get(0)\n\ndf_seizure = pd.read_parquet(BASE_PATH + 'train_eegs/266631836.parquet')\ndf_seizure.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T05:50:44.977481Z","iopub.execute_input":"2025-05-02T05:50:44.977796Z","iopub.status.idle":"2025-05-02T05:50:45.332588Z","shell.execute_reply.started":"2025-05-02T05:50:44.977773Z","shell.execute_reply":"2025-05-02T05:50:45.331642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Code for Spectrogram of Seizure T3 Data for 1 Participant\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom scipy.signal import spectrogram\n\nchannel_name = 'T3' \nsignal = df_seizure[channel_name].dropna().values  \nfs = 200  \n\nfrequencies, times, Sxx = spectrogram(signal, fs=fs, nperseg=256, noverlap=128)\n\nplt.figure(figsize=(10, 4))\nplt.pcolormesh(times, frequencies, 10 * np.log10(Sxx + 1e-10), shading='gouraud')\nplt.title(f'Seizure Spectrogram of {channel_name}')\nplt.ylabel('Frequency [Hz]')\nplt.xlabel('Time [sec]')\nplt.colorbar(label='Intensity [dB]')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T05:50:45.333470Z","iopub.execute_input":"2025-05-02T05:50:45.333763Z","iopub.status.idle":"2025-05-02T05:50:45.910261Z","shell.execute_reply.started":"2025-05-02T05:50:45.333739Z","shell.execute_reply":"2025-05-02T05:50:45.909424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Code for LRDA and Seizure T3 Data Table\nimport numpy as np\nimport pandas as pd\n\neeg_data = {\n    '266631836': {'path': '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/266631836.parquet', 'label': 'Seizure'},\n    '2222924277': {'path': '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/2222924277.parquet', 'label': 'LRDA'},\n    '2894007647': {'path': '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/2894007647.parquet', 'label': 'Seizure'},\n    '722738444': {'path': '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/722738444.parquet', 'label': 'LRDA'},\n    '338161210': {'path': '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/338161210.parquet', 'label': 'LRDA'},\n    '2088807520': {'path': '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/2088807520.parquet', 'label': 'Seizure'},\n    '3030710864': {'path': '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/3030710864.parquet', 'label': 'Seizure'},\n    '3190279138': {'path': '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/3190279138.parquet', 'label': 'Seizure'},\n    '1844014178': {'path': '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1844014178.parquet', 'label': 'LRDA'},\n    '2622179549': {'path': '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/2622179549.parquet', 'label': 'LRDA'},\n}\n\nmin_length = None\nfor pid, info in eeg_data.items():\n    try:\n        df = pd.read_parquet(info['path'])\n        if 'T3' in df.columns:\n            length = len(df['T3'])\n            if min_length is None or length < min_length:\n                min_length = length\n    except Exception as e:\n        print(f\"Failed to load {pid}: {e}\")\n\nprint(f\"Minimum T3 length: {min_length}\")\n\ndata_rows = []\n\nfor pid, info in eeg_data.items():\n    try:\n        df = pd.read_parquet(info['path'])\n        if 'T3' not in df.columns:\n            print(f\"Skipping {pid} — no T3 column\")\n            continue\n\n        T3_values = df['T3'].values[:min_length]\n        row = {\n            'Patient IDs': pid,\n            'Labels': info['label']\n        }\n        for t in range(min_length):\n            row[f'Time {t}'] = T3_values[t]\n\n        data_rows.append(row)\n    except Exception as e:\n        print(f\"Skipping {pid} due to error: {e}\")\n\nfinal_df = pd.DataFrame(data_rows)\n\nprint(f\"Final table shape: {final_df.shape}\")\nfinal_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T05:50:45.912599Z","iopub.execute_input":"2025-05-02T05:50:45.912920Z","iopub.status.idle":"2025-05-02T05:50:46.422611Z","shell.execute_reply.started":"2025-05-02T05:50:45.912897Z","shell.execute_reply":"2025-05-02T05:50:46.421653Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Code for PCA Table of Seizure and LRDA T3 Data\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\n\nfeatures_only = final_df.drop(columns=['Patient IDs', 'Labels']).fillna(final_df.mean(numeric_only=True))\n\nscaler = StandardScaler()\nscaled_features = scaler.fit_transform(features_only)\n\npca2 = PCA(n_components=2)\nprincipal_components = pca2.fit_transform(scaled_features)\n\npca2_df = pd.DataFrame(principal_components, columns=['PC3', 'PC4'])\npca2_df['Patient IDs'] = final_df['Patient IDs'].values\npca2_df = pca2_df.set_index('Patient IDs')\npca2_df['Labels'] = final_df['Labels'].values\n\npca2_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T05:50:46.423611Z","iopub.execute_input":"2025-05-02T05:50:46.423909Z","iopub.status.idle":"2025-05-02T05:50:50.123560Z","shell.execute_reply.started":"2025-05-02T05:50:46.423887Z","shell.execute_reply":"2025-05-02T05:50:50.122843Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Code for Spectrogram of PCA T3 Seizure v. LRDA data\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nsns.set(style='whitegrid')\n\nplt.figure(figsize=(10, 7))\nsns.scatterplot(\n    data=pca2_df,\n    x='PC3',\n    y='PC4',\n    hue='Labels',\n    palette={'Seizure': 'red', 'LRDA': 'blue'},\n    s=100,\n    edgecolor='black'\n)\n\nplt.title('PCA of EEG Data (T3) - Seizure vs LRDA', fontsize=16)\nplt.xlabel('Principal Component 3', fontsize=12)\nplt.ylabel('Principal Component 4', fontsize=12)\nplt.legend(title= 'Labels')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T05:50:50.124192Z","iopub.execute_input":"2025-05-02T05:50:50.124448Z","iopub.status.idle":"2025-05-02T05:50:50.541610Z","shell.execute_reply.started":"2025-05-02T05:50:50.124418Z","shell.execute_reply":"2025-05-02T05:50:50.540686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.preprocessing import LabelEncoder\n\nle = LabelEncoder()\ny_encoded = le.fit_transform(pca2_df['Labels']) \n\nX = pca2_df[['PC3', 'PC4']].values\nX = np.array(X).astype(float)\n\nclf = DecisionTreeClassifier(random_state=42)\nclf.fit(X, y_encoded)\n\nx_min, x_max = X[:, 0].min() - 1, X[:, 0].max() + 1\ny_min, y_max = X[:, 1].min() - 1, X[:, 1].max() + 1\nxx, yy = np.meshgrid(np.linspace(x_min - 35, x_max, 300),\n                     np.linspace(y_min - 30, y_max, 300))\n\nZ = clf.predict(np.c_[xx.ravel(), yy.ravel()])\nZ = Z.reshape(xx.shape)\n\nplt.figure(figsize=(10, 7))\nplt.contourf(xx, yy, Z, alpha=0.3, cmap='coolwarm')\nsns.set_style(\"white\")\nsns.scatterplot(\n    x=X[:, 0], y=X[:, 1], hue=pca2_df['Labels'],\n    palette={'Seizure': 'red', 'LRDA': 'blue'},\n    edgecolor='black', s=100\n)\n\nn_random_points = 20\nrandom_points = np.random.uniform(\n    low=(x_min, y_min),\n    high=(x_max, y_max),\n    size=(n_random_points, 2)\n)\n\nrandom_preds = clf.predict(random_points)\npredicted_labels = le.inverse_transform(random_preds)\ncolor_map = {'Seizure': 'red', 'LRDA': 'blue'}\n\nfor i, (point, label) in enumerate(zip(random_points, predicted_labels)):\n    plt.scatter(point[0], point[1], \n                color=color_map[label], \n                s=150, \n                edgecolor='black', \n                marker='X',\n                label='Random Test Point' if i == 0 else \"\"\n               )\n\nplt.title(\"Decision Boundary: PCA Components (Seizure vs LRDA) T3\", fontsize=16)\nplt.xlabel(\"PC3\", fontsize=12)\nplt.ylabel(\"PC4\", fontsize=12)\nplt.legend()\nplt.grid(False)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T05:50:50.542598Z","iopub.execute_input":"2025-05-02T05:50:50.543105Z","iopub.status.idle":"2025-05-02T05:50:51.157574Z","shell.execute_reply.started":"2025-05-02T05:50:50.543075Z","shell.execute_reply":"2025-05-02T05:50:51.156504Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}