{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\n\n# Load the train.csv metadata\ntrain = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\n\n# Display the first few rows of the dataset\nprint(\"First few rows of the dataset:\")\nprint(train.head())\n\n# Display the entire dataset shape\nprint(\"\\nDataset shape:\", train.shape)\n\n# Show the columns in the dataset\nprint(\"\\nColumns in the dataset:\")\nprint(train.columns)\n\n# Display normal patients (label: 'Other')\nnormal_patients_df = train[train['expert_consensus'] == 'Other']\nprint(\"\\nNormal patients in the dataset:\")\nprint(normal_patients_df)\n\n# Display the number of normal patients\nprint(\"\\nNumber of normal patients:\", len(normal_patients_df))\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:17.954603Z","iopub.execute_input":"2025-01-03T15:20:17.955019Z","iopub.status.idle":"2025-01-03T15:20:18.739332Z","shell.execute_reply.started":"2025-01-03T15:20:17.95497Z","shell.execute_reply":"2025-01-03T15:20:18.73798Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport matplotlib.pyplot as plt\nimport pyarrow.parquet as pq\nimport librosa\nimport glob\n\n# Load the train.csv metadata\ntrain = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\n\n# Define categories for normal and harmful brain activity\nnormal_labels = ['Other']  # Modify based on dataset labels for normal activity\nharmful_labels = ['Seizure', 'LPD', 'GPD', 'LRDA', 'GRDA']  # Labels for harmful activity\n\n# Filter patients based on labels\nnormal_patients = train[train['expert_consensus'].isin(normal_labels)]['patient_id'].unique()\nharmful_patients = train[train['expert_consensus'].isin(harmful_labels)]['patient_id'].unique()\n\n# Get available EEG files\neeg_path = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\navailable_eeg_files = glob.glob(eeg_path + \"*.parquet\")\navailable_eeg_ids = {os.path.basename(f).replace('.parquet', '') for f in available_eeg_files}\n\n# Print available EEG files\nprint(f\"Available EEG files: {len(available_eeg_files)}\")\nfor file in available_eeg_files[:10]:  # Print the first 10 available files\n    print(os.path.basename(file))\n\n# Find and print missing patient IDs\nunique_patient_ids = train['eeg_id'].unique()\nmissing_patients = set(unique_patient_ids) - available_eeg_ids\n\n# Calculate counts\nmissing_count = len(missing_patients)\navailable_count = len(available_eeg_ids)\n\nprint(f\"Missing patient IDs (not found in available EEG files): {missing_count}\")\nprint(f\"Available patient IDs: {available_count}\")\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:18.741432Z","iopub.execute_input":"2025-01-03T15:20:18.741843Z","iopub.status.idle":"2025-01-03T15:20:19.778642Z","shell.execute_reply.started":"2025-01-03T15:20:18.74179Z","shell.execute_reply":"2025-01-03T15:20:19.777123Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport matplotlib.pyplot as plt\nimport pyarrow.parquet as pq\nimport librosa\nimport glob\n\n# Load the train.csv metadata\ntrain = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\n\n# Define categories for normal brain activity\nnormal_labels = ['Other']  # Modify based on dataset labels for normal activity\n\n# Filter normal patients based on labels\nnormal_patients = train[train['expert_consensus'].isin(normal_labels)]['patient_id'].unique()\n\n# Get available EEG files\neeg_path = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\navailable_eeg_files = glob.glob(eeg_path + \"*.parquet\")\navailable_eeg_ids = {os.path.basename(f).replace('.parquet', '') for f in available_eeg_files}\n\n# Function to plot EEG waveforms for a patient\ndef plot_eeg_waveform(patient_id):\n    eeg_file = f'{eeg_path}{patient_id}.parquet'\n    \n    if os.path.exists(eeg_file):\n        eeg_data = pd.read_parquet(eeg_file)\n        \n        # Assuming the EEG data has columns for different channels\n        plt.figure(figsize=(12, 6))\n        plt.title(f'EEG Waveform for Patient {patient_id}')\n        \n        # Plotting the first few channels (modify based on actual columns)\n        for col in eeg_data.columns[:5]:  # Change 5 to the number of channels you want to visualize\n            plt.plot(eeg_data[col], label=col)\n        \n        plt.xlabel('Samples')\n        plt.ylabel('EEG Amplitude')\n        plt.legend()\n        plt.grid()\n        plt.show()\n    else:\n        print(f\"EEG file for patient {patient_id} does not exist.\")\n\n# Visualize waveforms for 2 normal patients\nfor patient_id in normal_patients[:2]:  # Limit to 2 normal patients\n    if str(patient_id) in available_eeg_ids:\n        plot_eeg_waveform(patient_id)\n    else:\n        print(f\"EEG file for patient {patient_id} not found. Skipping...\")\n\n# Here, you can proceed with your training data setup\n# (Add your training code here)\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:19.780411Z","iopub.execute_input":"2025-01-03T15:20:19.780925Z","iopub.status.idle":"2025-01-03T15:20:19.999444Z","shell.execute_reply.started":"2025-01-03T15:20:19.780874Z","shell.execute_reply":"2025-01-03T15:20:19.998217Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport matplotlib.pyplot as plt\nimport glob\n\n# Load the train.csv metadata\ntrain = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\n\n# Check unique expert consensus labels in the dataset\nunique_labels = train['expert_consensus'].unique()\nprint(\"Unique labels in expert_consensus:\", unique_labels)\n\n# Define categories for normal and harmful brain activity\nnormal_labels = ['Other']  # Modify based on dataset labels for normal activity\nharmful_labels = ['Seizure', 'LPD', 'GPD', 'LRDA', 'GRDA']  # Labels for harmful activity\n\n# Filter normal patients based on labels\nnormal_patients = train[train['expert_consensus'].isin(normal_labels)]['patient_id'].unique()\n\n# Get available EEG files\neeg_path = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\navailable_eeg_files = glob.glob(eeg_path + \"*.parquet\")\navailable_eeg_ids = {os.path.basename(f).replace('.parquet', '') for f in available_eeg_files}\n\n# Print available EEG files\nprint(f\"Available EEG files: {len(available_eeg_files)}\")\nfor file in available_eeg_files[:10]:  # Print the first 10 available files\n    print(os.path.basename(file))\n\n# Check how many normal patients exist\nprint(f\"Number of normal patients in dataset before filtering: {len(normal_patients)}\")\n\n# Identify available normal patients\navailable_normal_patients = [pid for pid in normal_patients if str(pid) in available_eeg_ids]\n\n# Check how many available normal patients there are\nprint(f\"Number of available normal patients: {len(available_normal_patients)}\")\n\n# Visualize waveforms for up to 2 available normal patients\nfor patient_id in available_normal_patients[:2]:  # Limit to 2 normal patients\n    plot_eeg_waveform(patient_id)\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:20.002537Z","iopub.execute_input":"2025-01-03T15:20:20.003019Z","iopub.status.idle":"2025-01-03T15:20:20.22534Z","shell.execute_reply.started":"2025-01-03T15:20:20.002971Z","shell.execute_reply":"2025-01-03T15:20:20.223872Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport matplotlib.pyplot as plt\nimport librosa\nimport glob\n\n# Load the train.csv metadata\ntrain = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\n\n# Define normal and harmful labels\nnormal_labels = ['Other']  \nharmful_labels = ['Seizure', 'LPD', 'GPD', 'LRDA', 'GRDA']\n\n# Filter for normal patients\nnormal_patients = train[train['expert_consensus'].isin(normal_labels)]['patient_id'].unique()\n\n# Get available EEG files\neeg_path = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\navailable_eeg_files = glob.glob(eeg_path + \"*.parquet\")\navailable_eeg_ids = {os.path.basename(f).replace('.parquet', '') for f in available_eeg_files}\n\n# Function to generate spectrograms\ndef spectrogram_from_eeg(parquet_path):\n    eeg = pd.read_parquet(parquet_path)\n    middle = (len(eeg) - 10_000) // 2\n    eeg = eeg.iloc[middle:middle + 10_000]\n\n    img = np.zeros((128, 256, 4), dtype='float32')\n\n    for k in range(4):  # Assuming 4 montage differences\n        COLS = FEATS[k]\n        for kk in range(4):\n            x = eeg[COLS[kk]].values - eeg[COLS[kk + 1]].values\n            x = np.nan_to_num(x, nan=np.nanmean(x))\n\n            mel_spec = librosa.feature.melspectrogram(\n                y=x, sr=200, hop_length=len(x) // 256,\n                n_fft=1024, n_mels=128, fmin=0, fmax=20, win_length=128)\n\n            mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max)[:, :(mel_spec.shape[1] // 32) * 32]\n            mel_spec_db = (mel_spec_db + 40) / 40\n            \n            img[:, :, k] += mel_spec_db\n\n        img[:, :, k] /= 4.0\n    \n    return img\n\n# Process available normal patients\nfor patient_id in normal_patients:\n    if str(patient_id) in available_eeg_ids:\n        eeg_file = f'{eeg_path}{patient_id}.parquet'\n        img = spectrogram_from_eeg(eeg_file)\n        # Save or display the spectrogram if needed\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:20.227208Z","iopub.execute_input":"2025-01-03T15:20:20.227727Z","iopub.status.idle":"2025-01-03T15:20:20.445187Z","shell.execute_reply.started":"2025-01-03T15:20:20.227676Z","shell.execute_reply":"2025-01-03T15:20:20.444008Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers, models\n\n# Define a simple CNN model\ndef create_model():\n    model = models.Sequential()\n    model.add(layers.Input(shape=(128, 256, 4)))  # Specify input shape explicitly\n    model.add(layers.Conv2D(32, (3, 3), activation='relu'))\n    model.add(layers.MaxPooling2D((2, 2)))\n    model.add(layers.Conv2D(64, (3, 3), activation='relu'))\n    model.add(layers.MaxPooling2D((2, 2)))\n    model.add(layers.Flatten())\n    model.add(layers.Dense(64, activation='relu'))\n    model.add(layers.Dense(6, activation='softmax'))  # Adjust for number of classes\n\n    model.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n    return model\n\n# Create and summarize the model\nmodel = create_model()\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:20.446716Z","iopub.execute_input":"2025-01-03T15:20:20.447166Z","iopub.status.idle":"2025-01-03T15:20:35.101558Z","shell.execute_reply.started":"2025-01-03T15:20:20.447119Z","shell.execute_reply":"2025-01-03T15:20:35.100375Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Number of available EEG files: {len(available_eeg_ids)}\")\nprint(f\"Number of normal patients to process: {len(normal_patients)}\")\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:35.103085Z","iopub.execute_input":"2025-01-03T15:20:35.103907Z","iopub.status.idle":"2025-01-03T15:20:35.110692Z","shell.execute_reply.started":"2025-01-03T15:20:35.103849Z","shell.execute_reply":"2025-01-03T15:20:35.109497Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for patient_id, img in all_eegs.items():\n    X.append(img)\n    label = train[train['patient_id'] == patient_id]['expert_consensus'].values[0]\n    print(f\"Patient {patient_id} label: {label}\")  # Debugging print\n    if label == 'Other':\n        y.append(0)  # Normal label\n    elif label in harmful_labels:\n        y.append(1)  # Harmful label\n    else:\n        print(f\"Unknown label for patient {patient_id}: {label}\")\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:35.112161Z","iopub.execute_input":"2025-01-03T15:20:35.112594Z","iopub.status.idle":"2025-01-03T15:20:35.886219Z","shell.execute_reply.started":"2025-01-03T15:20:35.112545Z","shell.execute_reply":"2025-01-03T15:20:35.883616Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 1: Identify Missing Patient IDs\nmissing_patient_ids = [pid for pid in normal_patients if str(pid) not in available_eeg_ids]\nprint(f\"Number of missing patient IDs: {len(missing_patient_ids)}\")\nprint(\"Missing patient IDs:\", missing_patient_ids[:10])  # Print the first 10 missing IDs\n\n# Step 2: Check Available EEG File Names\nprint(f\"Available EEG files: {len(available_eeg_files)}\")\nfor file in available_eeg_files[:10]:  # Print the first 10 available files\n    print(os.path.basename(file))\n\n# Step 3: Filter Train Data for Available EEG IDs\navailable_patient_ids = train[train['patient_id'].isin(available_eeg_ids)]['patient_id'].unique()\nprint(f\"Number of available patients in train dataset: {len(available_patient_ids)}\")\nprint(\"Available patient IDs in train dataset:\", available_patient_ids[:10])  # Print the first 10 available IDs\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:35.887379Z","iopub.status.idle":"2025-01-03T15:20:35.887844Z","shell.execute_reply.started":"2025-01-03T15:20:35.8876Z","shell.execute_reply":"2025-01-03T15:20:35.887651Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 1: Extract patient IDs from available EEG files\navailable_patient_ids_from_eeg = {int(os.path.basename(f).replace('.parquet', '')) for f in available_eeg_files}\nprint(f\"Number of patient IDs in available EEG files: {len(available_patient_ids_from_eeg)}\")\nprint(\"Available patient IDs from EEG files:\", list(available_patient_ids_from_eeg)[:10])  # Print first 10\n\n# Step 2: Check labels for these available patient IDs in train DataFrame\navailable_labels = train[train['patient_id'].isin(available_patient_ids_from_eeg)]\nprint(f\"Number of available labels for patients in EEG files: {len(available_labels)}\")\nprint(available_labels.head())  # Display the first few rows of available labels\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:35.889764Z","iopub.status.idle":"2025-01-03T15:20:35.890143Z","shell.execute_reply.started":"2025-01-03T15:20:35.88997Z","shell.execute_reply":"2025-01-03T15:20:35.889988Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Strip leading zeros from the patient_id column in train.csv and convert to integer\ntrain['patient_id'] = train['patient_id'].astype(str).str.lstrip('0').astype(int)\n\n# Check if this resolves the issue\navailable_labels = train[train['patient_id'].isin(available_patient_ids_from_eeg)]\nprint(f\"Number of available labels for patients in EEG files after stripping zeros: {len(available_labels)}\")\nprint(available_labels.head())\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:35.891265Z","iopub.status.idle":"2025-01-03T15:20:35.891687Z","shell.execute_reply.started":"2025-01-03T15:20:35.891482Z","shell.execute_reply":"2025-01-03T15:20:35.891501Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Try matching based on eeg_id\navailable_eeg_ids_from_train = train['eeg_id'].astype(str).unique()\navailable_eeg_ids_in_files = {os.path.basename(f).replace('.parquet', '') for f in available_eeg_files}\n\n# Check if any eeg_ids match\nmatching_eeg_ids = set(available_eeg_ids_from_train).intersection(available_eeg_ids_in_files)\nprint(f\"Number of matching eeg_ids: {len(matching_eeg_ids)}\")\n\n# Filter the train data based on matching eeg_ids\nmatching_labels = train[train['eeg_id'].isin(matching_eeg_ids)]\nprint(f\"Number of available labels for matching eeg_ids: {len(matching_labels)}\")\nprint(matching_labels.head())\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:35.893193Z","iopub.status.idle":"2025-01-03T15:20:35.893605Z","shell.execute_reply.started":"2025-01-03T15:20:35.8934Z","shell.execute_reply":"2025-01-03T15:20:35.893419Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport matplotlib.pyplot as plt\nimport pyarrow.parquet as pq\nimport glob\n\n# Path to available EEG files\neeg_path = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\n\n# Load the train metadata\ntrain = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\n\n# Define categories for normal and harmful brain activity\nnormal_labels = ['Other']  # Modify based on dataset labels for normal activity\nharmful_labels = ['Seizure', 'LPD', 'GPD', 'LRDA', 'GRDA']  # Labels for harmful activity\n\n# Filter patients based on labels\nnormal_patients = train[train['expert_consensus'].isin(normal_labels)]['patient_id'].unique()\nharmful_patients = train[train['expert_consensus'].isin(harmful_labels)]['patient_id'].unique()\n\n# Get available EEG files\navailable_eeg_files = glob.glob(eeg_path + \"*.parquet\")\n\n# Function to plot EEG waveforms\ndef plot_eeg_waveform(patient_id, display=True):\n    eeg_file = f'{eeg_path}{patient_id}.parquet'\n    \n    if os.path.exists(eeg_file):\n        eeg_data = pq.read_table(eeg_file).to_pandas()\n        middle = (len(eeg_data) - 10_000) // 2\n        eeg_segment = eeg_data.iloc[middle:middle + 10_000]  # Middle 50 seconds\n\n        # Plot the first available EEG channel (adjust column names as necessary)\n        plt.figure(figsize=(12, 6))\n        plt.plot(eeg_segment['Fp1'], label='Fp1')  # Replace 'Fp1' with the actual channel name\n        plt.title(f'EEG Waveform for Patient {patient_id}')\n        plt.xlabel('Samples')\n        plt.ylabel('Amplitude')\n        plt.legend()\n        plt.grid()\n        plt.show()\n    else:\n        print(f\"EEG file for patient {patient_id} not found.\")\n\n# Visualize EEG waveforms for a few normal patients\nfor patient_id in normal_patients[:2]:  # Adjust the number to visualize more\n    plot_eeg_waveform(patient_id)\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:35.895453Z","iopub.status.idle":"2025-01-03T15:20:35.895975Z","shell.execute_reply.started":"2025-01-03T15:20:35.895666Z","shell.execute_reply":"2025-01-03T15:20:35.895688Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport matplotlib.pyplot as plt\nimport pyarrow.parquet as pq\nimport glob\n\n# Path to available EEG files\neeg_path = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\n\n# Load the train metadata\ntrain = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\n\n# Define categories for normal brain activity\nnormal_labels = ['Other']  # Modify based on dataset labels for normal activity\n\n# Filter patients based on labels\nnormal_patients = train[train['expert_consensus'].isin(normal_labels)]['patient_id'].unique()\n\n# Get available EEG files\navailable_eeg_files = glob.glob(eeg_path + \"*.parquet\")\n\n# Create a set of available patient IDs from EEG files\navailable_patient_ids_from_eeg = {int(os.path.basename(f).replace('.parquet', '')) for f in available_eeg_files}\n\n# Function to plot EEG waveforms\ndef plot_eeg_waveform(patient_id):\n    eeg_file = f'{eeg_path}{patient_id}.parquet'\n    \n    if os.path.exists(eeg_file):\n        eeg_data = pq.read_table(eeg_file).to_pandas()\n        middle = (len(eeg_data) - 10_000) // 2\n        eeg_segment = eeg_data.iloc[middle:middle + 10_000]  # Middle 50 seconds\n\n        # Plot the first available EEG channel (adjust column names as necessary)\n        plt.figure(figsize=(12, 6))\n        plt.plot(eeg_segment['Fp1'], label='Fp1')  # Replace 'Fp1' with the actual channel name\n        plt.title(f'EEG Waveform for Patient {patient_id}')\n        plt.xlabel('Samples')\n        plt.ylabel('Amplitude')\n        plt.legend()\n        plt.grid()\n        plt.show()\n    else:\n        print(f\"EEG file for patient {patient_id} not found.\")\n\n# Visualize EEG waveforms for normal patients that have EEG files available\nfor patient_id in normal_patients:\n    if patient_id in available_patient_ids_from_eeg:\n        plot_eeg_waveform(patient_id)\n    else:\n        print(f\"EEG file for patient {patient_id} not found. Skipping...\")\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:35.897475Z","iopub.status.idle":"2025-01-03T15:20:35.897896Z","shell.execute_reply.started":"2025-01-03T15:20:35.897703Z","shell.execute_reply":"2025-01-03T15:20:35.897724Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.layers import Input, Dense, Dropout, LSTM, GRU, Flatten, Concatenate\nfrom tensorflow.keras.applications import EfficientNetB0, EfficientNetB2\nfrom tensorflow.keras.models import Model\n\n# Define the model architecture\n\n# Spectrogram input branch\nspectrogram_input = Input(shape=(128, 256, 3), name='spectrogram_input')  # Adjust shape as per spectrogram size\nefficientnet_block = EfficientNetB0(include_top=False, weights='imagenet')(spectrogram_input)\nspectrogram_features = Flatten()(efficientnet_block)\n\n# Raw EEG input branch\neeg_input = Input(shape=(10000, 1), name='eeg_input')  # Assuming 10,000 time steps\nlstm_block = LSTM(128, return_sequences=False)(eeg_input)  # or use GRU\neeg_features = Dense(128, activation='relu')(lstm_block)\n\n# Concatenate both feature branches\ncombined_features = Concatenate()([spectrogram_features, eeg_features])\n\n# Add dense layers with dropout\nx = Dense(256, activation='relu')(combined_features)\nx = Dropout(0.5)(x)\nx = Dense(128, activation='relu')(x)\nx = Dropout(0.5)(x)\n\n# Output layer for multi-class classification\noutput = Dense(6, activation='softmax', name='output')(x)  # 6 classes: SZ, LPD, GPD, LRDA, GRDA, other\n\n# Define the model\nmodel = Model(inputs=[spectrogram_input, eeg_input], outputs=output)\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='kullback_leibler_divergence', metrics=['accuracy'])\n\n# Model summary\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:35.899658Z","iopub.status.idle":"2025-01-03T15:20:35.900066Z","shell.execute_reply.started":"2025-01-03T15:20:35.899884Z","shell.execute_reply":"2025-01-03T15:20:35.899903Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport librosa\nimport os\nimport tensorflow as tf\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, Conv2D, MaxPooling2D, Flatten, Dense, Dropout, LSTM, GRU, concatenate\nfrom sklearn.model_selection import train_test_split\n\n# Load your dataset (e.g., train.csv)\ntrain_df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\n\n# Spectrogram Generation\ndef spectrogram_from_eeg(parquet_path):\n    eeg = pd.read_parquet(parquet_path)\n    middle = (len(eeg) - 10000) // 2\n    eeg = eeg.iloc[middle:middle+10000]\n    \n    img = np.zeros((128, 256, 4), dtype='float32')\n    \n    # Generate the spectrogram (from how-to-make-spectrogram-from-eeg)\n    # Use librosa for processing\n    for i in range(4):\n        x = eeg.iloc[:, i].values  # Example for generating a channel-specific spectrogram\n        mel_spec = librosa.feature.melspectrogram(y=x, sr=200, hop_length=len(x)//256, n_fft=1024, n_mels=128, fmin=0, fmax=20)\n        img[:, :, i] = librosa.power_to_db(mel_spec, ref=np.max)[:128, :256]\n    \n    return img\n\n# Generate Spectrograms and Raw EEG Data\nX_spectrograms, X_eeg, y = [], [], []\n\nfor index, row in train_df.iterrows():\n    eeg_file = f\"/path/to/eeg/{row['eeg_id']}.parquet\"\n    if os.path.exists(eeg_file):\n        # Generate spectrogram\n        spectrogram = spectrogram_from_eeg(eeg_file)\n        X_spectrograms.append(spectrogram)\n        \n        # Extract raw EEG data\n        eeg_data = pd.read_parquet(eeg_file)\n        raw_eeg = eeg_data['eeg_signal_column'].values  # Adjust column name based on your dataset\n        X_eeg.append(raw_eeg)\n        \n        # Append labels\n        y.append(row['expert_consensus'])\n\n# Convert to NumPy arrays\nX_spectrograms = np.array(X_spectrograms)\nX_eeg = np.array(X_eeg)\ny = np.array(y)\n\n# Split the dataset\nX_spectrogram_train, X_spectrogram_val, X_eeg_train, X_eeg_val, y_train, y_val = train_test_split(X_spectrograms, X_eeg, y, test_size=0.2, random_state=42)\n\n# Build EfficientNetB0 for Spectrograms\ninput_spectrogram = Input(shape=(128, 256, 4))\nefficient_net = tf.keras.applications.EfficientNetB0(include_top=False, weights='imagenet', input_tensor=input_spectrogram)\nflatten = Flatten()(efficient_net.output)\nspectrogram_features = Dense(64, activation='relu')(flatten)\n\n# Build LSTM/GRU for EEG Time-Series Data\ninput_eeg = Input(shape=(10000, 1))  # Adjust input shape as per raw EEG data\nlstm = LSTM(64)(input_eeg)\neeg_features = Dense(64, activation='relu')(lstm)\n\n# Concatenate the two branches\ncombined = concatenate([spectrogram_features, eeg_features])\n\n# Fully connected layers and output\nfc1 = Dense(64, activation='relu')(combined)\ndropout = Dropout(0.5)(fc1)\noutput = Dense(6, activation='softmax')(dropout)\n\n# Final model\nmodel = Model(inputs=[input_spectrogram, input_eeg], outputs=output)\nmodel.compile(optimizer='adam', loss='kl_divergence', metrics=['accuracy'])\n\n# Train the model\nhistory = model.fit([X_spectrogram_train, X_eeg_train], y_train, epochs=10, validation_data=([X_spectrogram_val, X_eeg_val], y_val))\n\n# Visualizing Training Results\nimport matplotlib.pyplot as plt\n\ndef plot_history(history):\n    plt.figure(figsize=(12, 4))\n    \n    # Accuracy plot\n    plt.subplot(1, 2, 1)\n    plt.plot(history.history['accuracy'], label='Train Accuracy')\n    plt.plot(history.history['val_accuracy'], label='Validation Accuracy')\n    plt.title('Accuracy')\n    plt.xlabel('Epochs')\n    plt.ylabel('Accuracy')\n    plt.legend()\n\n    # Loss plot\n    plt.subplot(1, 2, 2)\n    plt.plot(history.history['loss'], label='Train Loss')\n    plt.plot(history.history['val_loss'], label='Validation Loss')\n    plt.title('Loss')\n    plt.xlabel('Epochs')\n    plt.ylabel('Loss')\n    plt.legend()\n\n    plt.show()\n\nplot_history(history)\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:35.901542Z","iopub.status.idle":"2025-01-03T15:20:35.901947Z","shell.execute_reply.started":"2025-01-03T15:20:35.901768Z","shell.execute_reply":"2025-01-03T15:20:35.901788Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check if EEG files are found\neeg_files = []\n\nfor index, row in train_df.iterrows():\n    eeg_file = f\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/{row['eeg_id']}.parquet\"\n    if os.path.exists(eeg_file):\n        eeg_files.append(eeg_file)\n    else:\n        print(f\"EEG file for {row['eeg_id']} not found.\")\n\nprint(f\"Number of found EEG files: {len(eeg_files)}\")\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:35.902743Z","iopub.status.idle":"2025-01-03T15:20:35.903092Z","shell.execute_reply.started":"2025-01-03T15:20:35.90292Z","shell.execute_reply":"2025-01-03T15:20:35.902937Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load a sample EEG Parquet file to inspect the columns\nsample_eeg_file = eeg_files[0]  # Assuming `eeg_files` contains the paths to your EEG files\neeg_data = pd.read_parquet(sample_eeg_file)\n\n# Print the available columns in the file\nprint(eeg_data.columns)\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:35.904824Z","iopub.status.idle":"2025-01-03T15:20:35.905387Z","shell.execute_reply.started":"2025-01-03T15:20:35.905119Z","shell.execute_reply":"2025-01-03T15:20:35.905147Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport librosa\nimport matplotlib.pyplot as plt\n\ndef spectrogram_from_eeg(parquet_path, display=True):\n    # Load the EEG data from the parquet file\n    print(f\"Loading data from: {parquet_path}\")\n    try:\n        eeg = pd.read_parquet(parquet_path)\n    except Exception as e:\n        print(f\"Error loading data: {e}\")\n        return None\n    \n    print(f\"Data loaded. EEG shape: {eeg.shape}\")\n    \n    # Prepare a placeholder for the spectrogram (assuming 4 channels, adjust as needed)\n    img = np.zeros((128, 256, 4), dtype='float32')\n\n    for i in range(4):  # Process the first 4 columns of the EEG data\n        x = eeg.iloc[:, i].values\n        print(f\"Processing channel {i+1}, data shape: {x.shape}\")\n        \n        # Handle invalid values by replacing NaN/Inf with 0 or the mean\n        x = np.nan_to_num(x, nan=np.nanmean(x), posinf=0, neginf=0)\n        print(f\"After cleaning, data min: {np.min(x)}, max: {np.max(x)}\")\n        \n        # Check if the signal is valid (non-zero) before proceeding\n        if np.all(x == 0):\n            print(f\"Channel {i+1} has no valid data. Skipping...\")\n            continue\n\n        # Generate mel-spectrogram for the cleaned signal\n        mel_spec = librosa.feature.melspectrogram(\n            y=x, sr=200, hop_length=len(x)//256, n_fft=1024, n_mels=128, fmin=0, fmax=20)\n        img[:, :, i] = librosa.power_to_db(mel_spec, ref=np.max)[:128, :256]\n        \n        # Ensure the spectrogram is not empty or invalid\n        print(f\"Spectrogram shape for channel {i+1}: {img[:, :, i].shape}\")\n        if np.all(img[:, :, i] == 0):\n            print(f\"Spectrogram for channel {i+1} is empty. Check the input signal.\")\n        else:\n            if display:\n                plt.figure(figsize=(10, 4))\n                plt.imshow(img[:, :, i], aspect='auto', origin='lower')\n                plt.title(f'Spectrogram for Channel {i+1}')\n                plt.colorbar(format='%+2.0f dB')\n                plt.show()\n\n    print(\"Spectrogram generation completed.\")\n    return img\n\n# Example usage: \nparquet_file_path = \"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/2208063991.parquet\"\nspectrogram = spectrogram_from_eeg(parquet_file_path, display=True)\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:35.907535Z","iopub.status.idle":"2025-01-03T15:20:35.908125Z","shell.execute_reply.started":"2025-01-03T15:20:35.907836Z","shell.execute_reply":"2025-01-03T15:20:35.907865Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\ndef plot_eeg_signals(parquet_path):\n    # Load the EEG data from the parquet file\n    print(f\"Loading data from: {parquet_path}\")\n    try:\n        eeg = pd.read_parquet(parquet_path)\n    except Exception as e:\n        print(f\"Error loading data: {e}\")\n        return None\n\n    print(f\"Data loaded. EEG shape: {eeg.shape}\")\n    \n    # Plot EEG signals from the first 4 channels (you can adjust this as per your need)\n    plt.figure(figsize=(12, 6))\n    time_axis = np.arange(eeg.shape[0])  # Assuming each row corresponds to one time step\n\n    for i in range(4):  # Plot the first 4 columns of the EEG data\n        channel_data = eeg.iloc[:, i].values\n        # Handle NaN or Inf values\n        channel_data = np.nan_to_num(channel_data, nan=np.nanmean(channel_data), posinf=0, neginf=0)\n        \n        plt.plot(time_axis, channel_data + i * 100, label=f'Channel {i+1}')  # Offset signals for clarity\n\n    plt.title(\"EEG Signals (First 4 Channels)\")\n    plt.xlabel(\"Time\")\n    plt.ylabel(\"Amplitude (Offset)\")\n    plt.legend()\n    plt.grid(True)\n    plt.show()\n\n# Example usage:\nparquet_file_path = \"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/2208063991.parquet\"\nplot_eeg_signals(parquet_file_path)\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:35.910581Z","iopub.status.idle":"2025-01-03T15:20:35.911185Z","shell.execute_reply.started":"2025-01-03T15:20:35.910878Z","shell.execute_reply":"2025-01-03T15:20:35.910907Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\ndef plot_eeg_signals(parquet_path, zoom_factor=1.0, start_time=0, end_time=1000, dpi=100):\n    # Load the EEG data from the parquet file\n    print(f\"Loading data from: {parquet_path}\")\n    try:\n        eeg = pd.read_parquet(parquet_path)\n    except Exception as e:\n        print(f\"Error loading data: {e}\")\n        return None\n\n    print(f\"Data loaded. EEG shape: {eeg.shape}\")\n    \n    # Plot EEG signals from the first 4 channels (adjust as per your need)\n    plt.figure(figsize=(10 * zoom_factor, 6 * zoom_factor), dpi=dpi)\n    time_axis = np.arange(eeg.shape[0])  # Assuming each row corresponds to one time step\n    \n    # Set the time window for zooming\n    time_axis = time_axis[start_time:end_time]  # Zoom in on a specific time window\n    \n    for i in range(4):  # Plot the first 4 columns of the EEG data\n        channel_data = eeg.iloc[start_time:end_time, i].values  # Select the specific time window\n        # Handle NaN or Inf values\n        channel_data = np.nan_to_num(channel_data, nan=np.nanmean(channel_data), posinf=0, neginf=0)\n        \n        plt.plot(time_axis, channel_data + i * 50, label=f'Channel {i+1}')  # Smaller offset for zoomed signals\n\n    plt.title(f\"EEG Signals (Zoomed, Channels 1-4)\")\n    plt.xlabel(\"Time (Zoomed Range)\")\n    plt.ylabel(\"Amplitude (Offset)\")\n    plt.legend()\n    plt.grid(True)\n    plt.show()\n\n# Example usage with zoom factor, time range, and higher DPI:\nparquet_file_path = \"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/2208063991.parquet\"\nplot_eeg_signals(parquet_file_path, zoom_factor=1.5, start_time=500, end_time=1000, dpi=100)\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:35.912561Z","iopub.status.idle":"2025-01-03T15:20:35.913133Z","shell.execute_reply.started":"2025-01-03T15:20:35.91285Z","shell.execute_reply":"2025-01-03T15:20:35.912878Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\ndef plot_eeg_signals_before_after(pre_train_path, post_train_path, zoom_factor=1.5, start_time=0, end_time=1000, dpi=100):\n    # Load the EEG data from the pre-trained parquet file\n    print(f\"Loading pre-training data from: {pre_train_path}\")\n    try:\n        eeg_pre = pd.read_parquet(pre_train_path)\n    except Exception as e:\n        print(f\"Error loading pre-training data: {e}\")\n        return None\n    \n    print(f\"Pre-training data loaded. EEG shape: {eeg_pre.shape}\")\n\n    # Load the EEG data from the post-trained parquet file\n    print(f\"Loading post-training data from: {post_train_path}\")\n    try:\n        eeg_post = pd.read_parquet(post_train_path)\n    except Exception as e:\n        print(f\"Error loading post-training data: {e}\")\n        return None\n    \n    print(f\"Post-training data loaded. EEG shape: {eeg_post.shape}\")\n\n    # Set up the plot for both pre-training and post-training data\n    plt.figure(figsize=(12 * zoom_factor, 12 * zoom_factor), dpi=dpi)\n\n    # Time axis\n    time_pre = np.arange(eeg_pre.shape[0])[start_time:end_time]\n    time_post = np.arange(eeg_post.shape[0])[start_time:end_time]\n\n    # Plotting pre-training data\n    for i in range(4):  # Plot the first 4 channels for pre-training\n        channel_data = eeg_pre.iloc[start_time:end_time, i].values\n        channel_data = np.nan_to_num(channel_data, nan=np.nanmean(channel_data), posinf=0, neginf=0)\n        \n        plt.subplot(2, 1, 1)  # First row for pre-training\n        plt.plot(time_pre, channel_data + i * 50, label=f'Channel {i+1}')  # Smaller offset for clarity\n    \n    plt.title(\"EEG Signals Before Training (Channels 1-4)\")\n    plt.xlabel(\"Time (Zoomed Range)\")\n    plt.ylabel(\"Amplitude (Offset)\")\n    plt.legend()\n    plt.grid(True)\n\n    # Plotting post-training data\n    for i in range(4):  # Plot the first 4 channels for post-training\n        channel_data = eeg_post.iloc[start_time:end_time, i].values\n        channel_data = np.nan_to_num(channel_data, nan=np.nanmean(channel_data), posinf=0, neginf=0)\n\n        plt.subplot(2, 1, 2)  # Second row for post-training\n        plt.plot(time_post, channel_data + i * 50, label=f'Channel {i+1}')  # Smaller offset for clarity\n    \n    plt.title(\"EEG Signals After Training (Channels 1-4)\")\n    plt.xlabel(\"Time (Zoomed Range)\")\n    plt.ylabel(\"Amplitude (Offset)\")\n    plt.legend()\n    plt.grid(True)\n    \n    plt.tight_layout()\n    plt.show()\n\n# Example usage with paths for pre-trained and post-trained EEG data\npre_train_path = \"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/2208063991.parquet\"  # Change this to your pre-training file\npost_train_path = \"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1712674008.parquet\"  # Change this to your post-training file\n\nplot_eeg_signals_before_after(pre_train_path, post_train_path, zoom_factor=1.5, start_time=0, end_time=100, dpi=150)\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:35.914469Z","iopub.status.idle":"2025-01-03T15:20:35.9151Z","shell.execute_reply.started":"2025-01-03T15:20:35.914767Z","shell.execute_reply":"2025-01-03T15:20:35.914795Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\ndef plot_eeg_signals_before_after_discrete(pre_train_path, post_train_path, start_time=0, end_time=1000, dpi=100):\n    # Load the EEG data from the pre-trained parquet file\n    print(f\"Loading pre-training data from: {pre_train_path}\")\n    try:\n        eeg_pre = pd.read_parquet(pre_train_path)\n    except Exception as e:\n        print(f\"Error loading pre-training data: {e}\")\n        return None\n    \n    print(f\"Pre-training data loaded. EEG shape: {eeg_pre.shape}\")\n\n    # Load the EEG data from the post-trained parquet file\n    print(f\"Loading post-training data from: {post_train_path}\")\n    try:\n        eeg_post = pd.read_parquet(post_train_path)\n    except Exception as e:\n        print(f\"Error loading post-training data: {e}\")\n        return None\n    \n    print(f\"Post-training data loaded. EEG shape: {eeg_post.shape}\")\n\n    # Set up the plot for both pre-training and post-training data\n    plt.figure(figsize=(12, 12), dpi=dpi)\n\n    # Time axis\n    time_pre = np.arange(eeg_pre.shape[0])[start_time:end_time]\n    time_post = np.arange(eeg_post.shape[0])[start_time:end_time]\n\n    # Plotting pre-training data\n    for i in range(4):  # Plot the first 4 channels for pre-training\n        channel_data = eeg_pre.iloc[start_time:end_time, i].values\n        channel_data = np.nan_to_num(channel_data, nan=np.nanmean(channel_data), posinf=0, neginf=0)\n        \n        plt.subplot(2, 1, 1)  # First row for pre-training\n        plt.scatter(time_pre, channel_data + i * 50, label=f'Channel {i+1}', s=1)  # Use small dots for discrete visualization\n    \n    plt.title(\"EEG Signals Before Training (Discrete Representation)\")\n    plt.xlabel(\"Time (Zoomed Range)\")\n    plt.ylabel(\"Amplitude (Offset)\")\n    plt.legend()\n    plt.grid(True)\n\n    # Plotting post-training data\n    for i in range(4):  # Plot the first 4 channels for post-training\n        channel_data = eeg_post.iloc[start_time:end_time, i].values\n        channel_data = np.nan_to_num(channel_data, nan=np.nanmean(channel_data), posinf=0, neginf=0)\n\n        plt.subplot(2, 1, 2)  # Second row for post-training\n        plt.scatter(time_post, channel_data + i * 50, label=f'Channel {i+1}', s=1)  # Use small dots for discrete visualization\n    \n    plt.title(\"EEG Signals After Training (Discrete Representation)\")\n    plt.xlabel(\"Time (Zoomed Range)\")\n    plt.ylabel(\"Amplitude (Offset)\")\n    plt.legend()\n    plt.grid(True)\n    \n    plt.tight_layout()\n    plt.show()\n\n# Example usage with paths for pre-trained and post-trained EEG data\npre_train_path = \"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/2208063991.parquet\"  # Change this to your pre-training file\npost_train_path = \"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1712674008.parquet\"  # Change this to your post-training file\n\nplot_eeg_signals_before_after_discrete(pre_train_path, post_train_path, start_time=0, end_time=1000, dpi=100)\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:35.916752Z","iopub.status.idle":"2025-01-03T15:20:35.917307Z","shell.execute_reply.started":"2025-01-03T15:20:35.917039Z","shell.execute_reply":"2025-01-03T15:20:35.917068Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom scipy.signal import butter, filtfilt\n\ndef butterworth_filter(data, lowcut, highcut, fs, order=5):\n    nyq = 0.5 * fs\n    low = lowcut / nyq\n    high = highcut / nyq\n    b, a = butter(order, [low, high], btype='band')\n    filtered_data = filtfilt(b, a, data)\n    return filtered_data\n\ndef plot_eeg_signals_before_after(pre_train_path, post_train_path, fs=200, lowcut=0.5, highcut=40, start_time=0, end_time=1000, zoom_factor=1.0):\n    # Load the EEG data from the pre-trained parquet file\n    print(f\"Loading pre-training data from: {pre_train_path}\")\n    try:\n        eeg_pre = pd.read_parquet(pre_train_path)\n    except Exception as e:\n        print(f\"Error loading pre-training data: {e}\")\n        return None\n    \n    print(f\"Pre-training data loaded. EEG shape: {eeg_pre.shape}\")\n\n    # Load the EEG data from the post-trained parquet file\n    print(f\"Loading post-training data from: {post_train_path}\")\n    try:\n        eeg_post = pd.read_parquet(post_train_path)\n    except Exception as e:\n        print(f\"Error loading post-training data: {e}\")\n        return None\n    \n    print(f\"Post-training data loaded. EEG shape: {eeg_post.shape}\")\n\n    # Filter the EEG data\n    eeg_pre_filtered = eeg_pre.apply(lambda x: butterworth_filter(x, lowcut, highcut, fs), axis=0)\n    eeg_post_filtered = eeg_post.apply(lambda x: butterworth_filter(x, lowcut, highcut, fs), axis=0)\n\n    # Set up the plot for both pre-training and post-training data\n    plt.figure(figsize=(12 * zoom_factor, 12 * zoom_factor))\n\n    # Time axis\n    time_pre = np.arange(eeg_pre_filtered.shape[0])[start_time:end_time]\n    time_post = np.arange(eeg_post_filtered.shape[0])[start_time:end_time]\n\n    # Plotting pre-training data with step plot\n    plt.subplot(2, 1, 1)  # First row for pre-training\n    for i in range(4):  # Plot the first 4 channels for pre-training\n        channel_data = eeg_pre_filtered.iloc[start_time:end_time, i].values\n        plt.step(time_pre, channel_data + i * 50, label=f'Channel {i+1}')  # Smaller offset for clarity\n\n    plt.title(\"EEG Signals Before Training (Channels 1-4)\")\n    plt.xlabel(\"Time (seconds)\")\n    plt.ylabel(\"Amplitude (Offset)\")\n    plt.legend()\n    plt.grid(True)\n\n    # Plotting post-training data with step plot\n    plt.subplot(2, 1, 2)  # Second row for post-training\n    for i in range(4):  # Plot the first 4 channels for post-training\n        channel_data = eeg_post_filtered.iloc[start_time:end_time, i].values\n        plt.step(time_post, channel_data + i * 50, label=f'Channel {i+1}')  # Smaller offset for clarity\n\n    plt.title(\"EEG Signals After Training (Channels 1-4)\")\n    plt.xlabel(\"Time (seconds)\")\n    plt.ylabel(\"Amplitude (Offset)\")\n    plt.legend()\n    plt.grid(True)\n    \n    plt.tight_layout()\n    plt.show()\n\n# Example usage with paths for pre-trained and post-trained EEG data\npre_train_path = \"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/2208063991.parquet\"  # Change this to your pre-training file\npost_train_path = \"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1712674008.parquet\"  # Change this to your post-training file\n\nplot_eeg_signals_before_after(pre_train_path, post_train_path, start_time=0, end_time=1000)\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:35.919518Z","iopub.status.idle":"2025-01-03T15:20:35.920158Z","shell.execute_reply.started":"2025-01-03T15:20:35.919852Z","shell.execute_reply":"2025-01-03T15:20:35.919883Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom scipy.signal import butter, filtfilt\n\n# Butterworth filter for noise reduction\ndef butterworth_filter(data, lowcut, highcut, fs, order=5):\n    nyq = 0.5 * fs\n    low = lowcut / nyq\n    high = highcut / nyq\n    b, a = butter(order, [low, high], btype='band')\n    filtered_data = filtfilt(b, a, data)\n    return filtered_data\n\n# Function to plot EEG signals before and after training, using step plots\ndef plot_eeg_signals_step(pre_train_path, post_train_path, fs=200, lowcut=0.5, highcut=40, start_time=0, end_time=1000, zoom_factor=1.0):\n    # Load pre-training EEG data\n    eeg_pre = pd.read_parquet(pre_train_path)\n    \n    # Load post-training EEG data\n    eeg_post = pd.read_parquet(post_train_path)\n\n    # Filter the EEG data using Butterworth filter\n    eeg_pre_filtered = eeg_pre.apply(lambda x: butterworth_filter(x, lowcut, highcut, fs), axis=0)\n    eeg_post_filtered = eeg_post.apply(lambda x: butterworth_filter(x, lowcut, highcut, fs), axis=0)\n\n    # Set up the figure for plotting both pre and post-training data\n    plt.figure(figsize=(12 * zoom_factor, 12 * zoom_factor))\n\n    # Create time axis for pre and post data\n    time_pre = np.arange(eeg_pre_filtered.shape[0])[start_time:end_time]\n    time_post = np.arange(eeg_post_filtered.shape[0])[start_time:end_time]\n\n    # Plot pre-training data using step plot\n    plt.subplot(2, 1, 1)\n    for i in range(4):  # Plot the first 4 channels\n        channel_data = eeg_pre_filtered.iloc[start_time:end_time, i].values\n        plt.step(time_pre, channel_data + i * 50, label=f'Channel {i+1}', where='mid')  # Step plot with \"mid\" alignment\n\n    plt.title(\"Discrete EEG Signals Before Training (Step Plot)\")\n    plt.xlabel(\"Time (samples)\")\n    plt.ylabel(\"Amplitude (Offset)\")\n    plt.legend()\n    plt.grid(True)\n\n    # Plot post-training data using step plot\n    plt.subplot(2, 1, 2)\n    for i in range(4):  # Plot the first 4 channels\n        channel_data = eeg_post_filtered.iloc[start_time:end_time, i].values\n        plt.step(time_post, channel_data + i * 50, label=f'Channel {i+1}', where='mid')  # Step plot with \"mid\" alignment\n\n    plt.title(\"Discrete EEG Signals After Training (Step Plot)\")\n    plt.xlabel(\"Time (samples)\")\n    plt.ylabel(\"Amplitude (Offset)\")\n    plt.legend()\n    plt.grid(True)\n    \n    plt.tight_layout()\n    plt.show()\n\n# Example usage with paths to pre-trained and post-trained EEG data\npre_train_path = \"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/2208063991.parquet\"  # Pre-training file\npost_train_path = \"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1712674008.parquet\"  # Post-training file\n\n# Call the plotting function to visualize in a discrete way\nplot_eeg_signals_step(pre_train_path, post_train_path, start_time=500, end_time=1500)\n","metadata":{"execution":{"iopub.status.busy":"2025-01-03T15:20:54.718859Z","iopub.execute_input":"2025-01-03T15:20:54.719274Z","iopub.status.idle":"2025-01-03T15:20:56.161591Z","shell.execute_reply.started":"2025-01-03T15:20:54.719236Z","shell.execute_reply":"2025-01-03T15:20:56.16017Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\n\n# Assuming y_test and y_pred are already defined\nprint(\"Accuracy:\", accuracy_score(test, y_pred))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T15:27:19.760273Z","iopub.execute_input":"2025-01-03T15:27:19.760879Z","iopub.status.idle":"2025-01-03T15:27:19.798046Z","shell.execute_reply.started":"2025-01-03T15:27:19.760823Z","shell.execute_reply":"2025-01-03T15:27:19.796213Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.metrics import classification_report, accuracy_score\nfrom scipy.signal import butter, lfilter\nfrom sklearn.preprocessing import StandardScaler\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense\n\n# Load the dataset\ntrain = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\n\n# Function to apply a Butterworth filter\ndef butter_bandpass(lowcut, highcut, fs, order=5):\n    nyq = 0.5 * fs\n    low = lowcut / nyq\n    high = highcut / nyq\n    b, a = butter(order, [low, high], btype='band')\n    return b, a\n\ndef bandpass_filter(data, lowcut, highcut, fs=200, order=5):\n    b, a = butter_bandpass(lowcut, highcut, fs, order=order)\n    y = lfilter(b, a, data)\n    return y\n\n# Preprocessing: Filter EEG data\ndef preprocess_eeg(eeg_data):\n    filtered_data = bandpass_filter(eeg_data, 0.5, 30)\n    return filtered_data\n\n# Feature Extraction using CSP (simplified example)\ndef extract_csp_features(eeg_data):\n    features = np.array([np.mean(eeg_data), np.var(eeg_data)])\n    return features\n\n# Initialize lists for features and labels for CSP + KNN\nfeatures_list_knn = []\nlabels_list_knn = []\n\n# Iterate over each row in train DataFrame to load corresponding EEG data for KNN\nfor index, row in train.iterrows():\n    eeg_id = row['eeg_id']  \n    eeg_file_path = f'/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/eeg_id_{eeg_id}.csv'\n    \n    if os.path.isfile(eeg_file_path):\n        eeg_data = pd.read_csv(eeg_file_path)\n        eeg_signals = eeg_data.iloc[:, 1:].values  \n        \n        filtered_eeg = preprocess_eeg(eeg_signals)\n        features = extract_csp_features(filtered_eeg)\n        features_list_knn.append(features)\n        labels_list_knn.append(row['expert_consensus'])\n\n# Convert to numpy arrays for KNN\nX_knn = np.array(features_list_knn)\ny_knn = np.array(labels_list_knn)\n\n# Split the dataset into training and testing sets for KNN\nX_train_knn, X_test_knn, y_train_knn, y_test_knn = train_test_split(X_knn, y_knn, test_size=0.2, random_state=42)\n\n# Standardize features for KNN\nscaler_knn = StandardScaler()\nX_train_scaled_knn = scaler_knn.fit_transform(X_train_knn)\nX_test_scaled_knn = scaler_knn.transform(X_test_knn)\n\n# Initialize and train KNN classifier\nknn = KNeighborsClassifier(n_neighbors=5)  \nknn.fit(X_train_scaled_knn, y_train_knn)\n\n# Make predictions on the test set for KNN\ny_pred_knn = knn.predict(X_test_scaled_knn)\n\n# Evaluate the KNN model\nprint(\"KNN Accuracy:\", accuracy_score(y_test_knn, y_pred_knn))\nprint(\"KNN Classification Report:\\n\", classification_report(y_test_knn, y_pred_knn))\n\n# Now implement EfficientNet model for comparison\n\n# Prepare data for EfficientNet (assuming spectrograms are generated from EEG signals)\n# This is a placeholder; you need to generate spectrograms from your EEG data.\ndef create_spectrograms(eeg_signals):\n    # Placeholder function to convert EEG signals to spectrograms.\n    # You should implement the actual conversion logic here.\n    return np.random.rand(50, 50)  # Example: returning random data\n\nfeatures_list_effnet = []\nlabels_list_effnet = []\n\nfor index, row in train.iterrows():\n    eeg_id = row['eeg_id']  \n    eeg_file_path = f'/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/eeg_id_{eeg_id}.csv'\n    \n    if os.path.isfile(eeg_file_path):\n        eeg_data = pd.read_csv(eeg_file_path)\n        eeg_signals = eeg_data.iloc[:, 1:].values  \n        \n        spectrogram = create_spectrograms(eeg_signals)  # Generate spectrograms here.\n        features_list_effnet.append(spectrogram)\n        labels_list_effnet.append(row['expert_consensus'])\n\nX_effnet = np.array(features_list_effnet)\ny_effnet = np.array(labels_list_effnet)\n\n# Split the dataset into training and testing sets for EfficientNet\nX_train_effnet, X_test_effnet, y_train_effnet, y_test_effnet = train_test_split(X_effnet.reshape(-1, 50, 50, 1), y_effnet,\n                                                                                test_size=0.2,\n                                                                                random_state=42)\n\n# Define EfficientNet model (using a simplified CNN structure as an example)\nmodel_effnet = Sequential()\nmodel_effnet.add(Conv2D(32, (3, 3), activation='relu', input_shape=(50, 50, 1)))\nmodel_effnet.add(MaxPooling2D(pool_size=(2, 2)))\nmodel_effnet.add(Flatten())\nmodel_effnet.add(Dense(64, activation='relu'))\nmodel_effnet.add(Dense(len(np.unique(y_effnet)), activation='softmax'))\n\nmodel_effnet.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\n# Train EfficientNet model\nmodel_effnet.fit(X_train_effnet, y_train_effnet, epochs=10)\n\n# Evaluate EfficientNet model\neffnet_loss, effnet_accuracy = model_effnet.evaluate(X_test_effnet, y_test_effnet)\nprint(\"EfficientNet Accuracy:\", effnet_accuracy)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T15:34:50.541442Z","iopub.execute_input":"2025-01-03T15:34:50.541885Z","iopub.status.idle":"2025-01-03T15:35:14.247855Z","shell.execute_reply.started":"2025-01-03T15:34:50.541846Z","shell.execute_reply":"2025-01-03T15:35:14.246233Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.metrics import classification_report, accuracy_score\nfrom scipy.signal import butter, lfilter\nfrom sklearn.preprocessing import StandardScaler\nimport os\n\n# Load the dataset\ntrain = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nprint('Train shape:', train.shape)\ndisplay(train.head())\n\n# Function to apply a Butterworth filter\ndef butter_bandpass(lowcut, highcut, fs, order=5):\n    nyq = 0.5 * fs\n    low = lowcut / nyq\n    high = highcut / nyq\n    b, a = butter(order, [low, high], btype='band')\n    return b, a\n\ndef bandpass_filter(data, lowcut, highcut, fs=200, order=5):\n    b, a = butter_bandpass(lowcut, highcut, fs, order=order)\n    y = lfilter(b, a, data)\n    return y\n\n# Preprocessing: Filter EEG data\ndef preprocess_eeg(eeg_data):\n    filtered_data = bandpass_filter(eeg_data, 0.5, 30)\n    return filtered_data\n\n# Feature Extraction using CSP (simplified example)\ndef extract_csp_features(eeg_data):\n    features = np.array([np.mean(eeg_data), np.var(eeg_data)])\n    return features\n\n# Initialize lists for features and labels for CSP + KNN\nfeatures_list_knn = []\nlabels_list_knn = []\n\n# Iterate over each row in train DataFrame to load corresponding EEG data for KNN\nfor index, row in train.iterrows():\n    eeg_id = row['eeg_id']  \n    eeg_file_path = f'/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/eeg_id_{eeg_id}.csv'\n    \n    if os.path.isfile(eeg_file_path):\n        eeg_data = pd.read_csv(eeg_file_path)\n        eeg_signals = eeg_data.iloc[:, 1:].values  \n        \n        filtered_eeg = preprocess_eeg(eeg_signals)\n        features = extract_csp_features(filtered_eeg)\n        features_list_knn.append(features)\n        labels_list_knn.append(row['expert_consensus'])\n        \n        # Debugging output\n        print(f\"Processed EEG ID: {eeg_id}, Features: {features}, Label: {row['expert_consensus']}\")\n    else:\n        print(f\"File not found: {eeg_file_path}\")\n\n# Convert to numpy arrays for KNN\nX_knn = np.array(features_list_knn)\ny_knn = np.array(labels_list_knn)\n\n# Check if features were extracted successfully\nif X_knn.size == 0:\n    raise ValueError(\"No features extracted; please check the data loading process.\")\n\n# Split the dataset into training and testing sets for KNN\nX_train_knn, X_test_knn, y_train_knn, y_test_knn = train_test_split(X_knn, y_knn, test_size=0.2, random_state=42)\n\n# Standardize features for KNN\nscaler_knn = StandardScaler()\nX_train_scaled_knn = scaler_knn.fit_transform(X_train_knn)\nX_test_scaled_knn = scaler_knn.transform(X_test_knn)\n\n# Initialize and train KNN classifier\nknn = KNeighborsClassifier(n_neighbors=5)  \nknn.fit(X_train_scaled_knn, y_train_knn)\n\n# Make predictions on the test set for KNN\ny_pred_knn = knn.predict(X_test_scaled_knn)\n\n# Evaluate the KNN model\nprint(\"KNN Accuracy:\", accuracy_score(y_test_knn, y_pred_knn))\nprint(\"KNN Classification Report:\\n\", classification_report(y_test_knn, y_pred_knn))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T15:36:17.851174Z","iopub.execute_input":"2025-01-03T15:36:17.851577Z","execution_failed":"2025-01-03T16:24:14.243Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}