{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd\nimport matplotlib.pyplot as plt # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-12T13:55:17.416716Z","iopub.execute_input":"2025-05-12T13:55:17.417050Z","iopub.status.idle":"2025-05-12T13:56:20.274794Z","shell.execute_reply.started":"2025-05-12T13:55:17.417018Z","shell.execute_reply":"2025-05-12T13:56:20.273824Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ntrain_df = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")\nprint(train_df.head())\nprint(train_df.info())\nprint(train_df.describe())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T13:56:20.276132Z","iopub.execute_input":"2025-05-12T13:56:20.276540Z","iopub.status.idle":"2025-05-12T13:56:20.671533Z","shell.execute_reply.started":"2025-05-12T13:56:20.276516Z","shell.execute_reply":"2025-05-12T13:56:20.670651Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['expert_consensus'].value_counts().sort_index().plot(kind='bar', title=\"Class Distribution\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T13:56:20.672324Z","iopub.execute_input":"2025-05-12T13:56:20.672664Z","iopub.status.idle":"2025-05-12T13:56:21.081939Z","shell.execute_reply.started":"2025-05-12T13:56:20.672640Z","shell.execute_reply":"2025-05-12T13:56:21.081148Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pyarrow.parquet as pq\n\nsample_file = \"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/2208063991.parquet\"\neeg_df = pd.read_parquet(sample_file)\nprint(eeg_df.head())\nprint(eeg_df.columns)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T13:56:21.083952Z","iopub.execute_input":"2025-05-12T13:56:21.084191Z","iopub.status.idle":"2025-05-12T13:56:21.226790Z","shell.execute_reply.started":"2025-05-12T13:56:21.084174Z","shell.execute_reply":"2025-05-12T13:56:21.225748Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_patients = train_df['patient_id'].nunique()\nprint(f\"Number of unique patients in train dataset: {num_patients}\")\n\n# Number of unique EEG IDs\nnum_eeg_ids = train_df['eeg_id'].nunique()\nprint(f\"Number of unique EEG IDs in train dataset: {num_eeg_ids}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T13:56:21.228120Z","iopub.execute_input":"2025-05-12T13:56:21.228441Z","iopub.status.idle":"2025-05-12T13:56:21.239251Z","shell.execute_reply.started":"2025-05-12T13:56:21.228420Z","shell.execute_reply":"2025-05-12T13:56:21.238198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport matplotlib.pyplot as plt \nimport seaborn as sns\nnumerical_columns = train_df.select_dtypes(include=['int64', 'float64']).columns\ncorr_matrix = train_df[numerical_columns].corr()\n\n# Increase the size of the figure\nplt.figure(figsize=(12, 8)) \n\n# Heatmap for correlation analysis\n# We are using 'coolwarm' colormap here to distinguish positive and negative correlations easily\nsns.heatmap(corr_matrix, annot=True, cmap='coolwarm')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T13:56:21.240611Z","iopub.execute_input":"2025-05-12T13:56:21.240885Z","iopub.status.idle":"2025-05-12T13:56:22.022876Z","shell.execute_reply.started":"2025-05-12T13:56:21.240864Z","shell.execute_reply":"2025-05-12T13:56:22.021921Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.hist(figsize=(15, 10))\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T13:56:22.024467Z","iopub.execute_input":"2025-05-12T13:56:22.024836Z","iopub.status.idle":"2025-05-12T13:56:23.993195Z","shell.execute_reply.started":"2025-05-12T13:56:22.024809Z","shell.execute_reply":"2025-05-12T13:56:23.992229Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"eeg = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/2208063991.parquet')\n\n#List of columns to plot\ncolumns_to_plot = [\n    'Fp1', 'F3', 'C3', 'P3', 'F7', 'T3', 'T5', 'O1', 'Fz', 'Cz', 'Pz', \n    'Fp2', 'F4', 'C4', 'P4', 'F8', 'T4', 'T6', 'O2', 'EKG'\n]\n\n# Determine the number of rows/columns needed for subplots\nnum_plots = len(columns_to_plot)\nnum_columns = 2  # Set to 2 as per the previous code\nnum_rows = num_plots // num_columns + (num_plots % num_columns > 0)\n\n# Create subplots\nfig, axes = plt.subplots(num_rows, num_columns, figsize=(20, num_rows * 4))\n\n# Flatten the axes array for easy iteration\naxes = axes.flatten()\n\n# Plot each column in a subplot\nfor i, col in enumerate(columns_to_plot):\n    axes[i].plot(eeg[col])\n    axes[i].set_title(f'Electrode: {col}', fontsize=14)\n\n# Hide any unused subplots\nfor ax in axes[len(columns_to_plot):]:\n    ax.set_visible(False)\n\n# Set the overall figure title\nfig.suptitle('EEG Data Visualization Based on the International 10-20 System', fontsize=30, y=1.02)\n\n# Adjust layout to prevent overlap\nplt.tight_layout()\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T13:56:23.994117Z","iopub.execute_input":"2025-05-12T13:56:23.994417Z","iopub.status.idle":"2025-05-12T13:56:28.545729Z","shell.execute_reply.started":"2025-05-12T13:56:23.994395Z","shell.execute_reply":"2025-05-12T13:56:28.544707Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def visualize_eeg_signal(df, title, path=None):\n    eeg_columns = [\n        'Fp1', 'F3', 'C3', 'P3', 'F7', 'T3',\n        'T5', 'O1', 'Fz', 'Cz', 'Pz', 'Fp2',\n        'F4', 'C4', 'P4', 'F8', 'T4', 'T6',\n        'O2',\n    ]\n    ekg_column = 'EKG'\n    eeg_spacing = 500\n    \n    fig, axes = plt.subplots(figsize=(24, 24), nrows=2, height_ratios=[10, 1], dpi=100)\n\n    for column_idx, column in enumerate(eeg_columns):\n        axes[0].plot(np.arange(0, df.shape[0]), df[column] + (eeg_spacing * column_idx), linewidth=0.5, color='black')\n        \n    y_ticks = np.arange(0, len(eeg_columns)) * eeg_spacing - 100\n    axes[0].set_yticks(y_ticks)\n    axes[0].set_yticklabels(eeg_columns)\n    axes[0].tick_params(axis='x', labelsize=15)\n    axes[0].tick_params(axis='y', labelsize=15)\n    axes[0].set_xlabel('')\n    axes[0].set_ylabel('')\n    axes[0].set_title(title, size=15, pad=12.5, loc='center')\n    \n    axes[1].plot(np.arange(0, df.shape[0]), df['EKG'], linewidth=0.5, color='black')\n    axes[1].set_yticks(np.array(axes[1].get_yticks()) * 1.5)\n    axes[1].tick_params(axis='x', labelsize=12.5)\n    axes[1].tick_params(axis='y', labelsize=12.5)\n    axes[1].set_xlabel('')\n    axes[1].set_ylabel('')\n    axes[1].set_title('EKG', size=15, pad=12.5, loc='center')\n    \n    if path is None:\n        plt.show()\n    else:\n        plt.savefig(path, bbox_inches='tight')\n        plt.close(fig)\n\n\nvisualize_eeg_signal(\n    df=eeg,\n    title='EEG 2208063991'\n)\n\n        \n    \n   \n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T13:56:28.546922Z","iopub.execute_input":"2025-05-12T13:56:28.547196Z","iopub.status.idle":"2025-05-12T13:56:29.373987Z","shell.execute_reply.started":"2025-05-12T13:56:28.547176Z","shell.execute_reply":"2025-05-12T13:56:29.373021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\ntrain_df['expert_consensus'].value_counts().plot(kind='bar')\nplt.title(\"Distribution of Expert Consensus Labels\")\nplt.xticks(rotation=45)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T13:56:29.377201Z","iopub.execute_input":"2025-05-12T13:56:29.377889Z","iopub.status.idle":"2025-05-12T13:56:29.587263Z","shell.execute_reply.started":"2025-05-12T13:56:29.377834Z","shell.execute_reply":"2025-05-12T13:56:29.586558Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pyarrow.parquet as pq\nfrom scipy.signal import welch, spectrogram\nimport os\n\n# Config\nplt.style.use('ggplot')\nplt.rcParams['figure.figsize'] = (15, 6)\nsns.set_palette(\"husl\")\n\n# Load data\ntrain_df = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")\nlabels = [\"seizure_vote\", \"lpd_vote\", \"gpd_vote\", \"lrda_vote\", \"grda_vote\", \"other_vote\"]\nplt.figure(figsize=(14, 7))\ntrain_df[labels].sum().sort_values().plot(kind='barh')\nplt.title(\"Total Votes per Label Across All Samples\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T13:56:29.588201Z","iopub.execute_input":"2025-05-12T13:56:29.588525Z","iopub.status.idle":"2025-05-12T13:56:30.075958Z","shell.execute_reply.started":"2025-05-12T13:56:29.588498Z","shell.execute_reply":"2025-05-12T13:56:30.075075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from itertools import combinations\nfor combo in combinations(labels, 2):\n    plt.figure()\n    sns.scatterplot(data=train_df, x=combo[0], y=combo[1], alpha=0.6)\n    plt.title(f\"Vote Interaction: {combo[0]} vs {combo[1]}\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T13:56:30.076883Z","iopub.execute_input":"2025-05-12T13:56:30.077119Z","iopub.status.idle":"2025-05-12T13:56:37.245848Z","shell.execute_reply.started":"2025-05-12T13:56:30.077102Z","shell.execute_reply":"2025-05-12T13:56:37.244941Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(14, 8))\nsns.boxplot(data=train_df[labels])\nplt.title(\"Distribution of Expert Vote Confidence per Label\")\nplt.xticks(rotation=45)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T13:56:37.246931Z","iopub.execute_input":"2025-05-12T13:56:37.247266Z","iopub.status.idle":"2025-05-12T13:56:37.734777Z","shell.execute_reply.started":"2025-05-12T13:56:37.247243Z","shell.execute_reply":"2025-05-12T13:56:37.733761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_spectrograms_for_class(target_label, n_samples=3):\n    class_samples = train_df[train_df[target_label] > 0].sample(n_samples)\n    for _, row in class_samples.iterrows():\n        spec = pq.read_pandas(\n            f\"/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/{row.spectrogram_id}.parquet\"\n        ).to_pandas()\n        \n        plt.figure(figsize=(18, 4))\n        plt.imshow(spec.iloc[:, 1:].T, aspect='auto', cmap='viridis', vmax=5)\n        plt.colorbar(label='Power (dB)')\n        plt.title(f\"Spectrogram for {target_label} (Patient {row.patient_id}, Votes: {row[labels].to_dict()})\")\n        plt.xlabel(\"Time (bins)\")\n        plt.ylabel(\"Frequency (Hz)\")\n        plt.show()\n\nfor label in labels:\n    plot_spectrograms_for_class(label)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T13:56:37.735588Z","iopub.execute_input":"2025-05-12T13:56:37.735835Z","iopub.status.idle":"2025-05-12T13:56:47.876724Z","shell.execute_reply.started":"2025-05-12T13:56:37.735817Z","shell.execute_reply":"2025-05-12T13:56:47.875588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_eegs_for_class(target_label, n_samples=2, electrodes=('Fp1', 'Fp2', 'T3', 'T4')):\n    class_samples = train_df[train_df[target_label] > 0].sample(n_samples)\n    for _, row in class_samples.iterrows():\n        eeg = pq.read_pandas(\n            f\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/{row.eeg_id}.parquet\"\n        ).to_pandas()\n        \n        plt.figure(figsize=(18, 8))\n        for col in electrodes:\n            plt.plot(eeg[col].values[:2000], label=col)\n        plt.legend()\n        plt.title(f\"EEG for {target_label} (Patient {row.patient_id})\")\n        plt.xlabel(\"Time (samples)\")\n        plt.ylabel(\"Amplitude (µV)\")\n        plt.show()\n\nfor label in labels:\n    plot_eegs_for_class(label)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T13:56:47.878244Z","iopub.execute_input":"2025-05-12T13:56:47.878617Z","iopub.status.idle":"2025-05-12T13:56:53.380045Z","shell.execute_reply.started":"2025-05-12T13:56:47.878584Z","shell.execute_reply":"2025-05-12T13:56:53.379092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport pyarrow.parquet as pq\nimport matplotlib.pyplot as plt\nimport random\n\n# Directory where EEG parquet files are stored\nEEG_DIR = \"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs\"\n\n# Get list of all available parquet files\neeg_files = [f for f in os.listdir(EEG_DIR) if f.endswith(\".parquet\")]\n\n# Randomly select 5 files (or any number you prefer)\nsample_files = random.sample(eeg_files, 5)\n\n# Loop over each selected file and plot signals from specific electrodes\nfor file in sample_files:\n    eeg_id = file.split(\".\")[0]\n    eeg_path = os.path.join(EEG_DIR, file)\n\n    # Load EEG data\n    eeg_data = pd.read_parquet(eeg_path)\n\n    # Electrodes to plot\n    electrodes_to_plot = ['Fp1', 'F3', 'C3', 'O1']  # you can change this list\n\n    plt.figure(figsize=(14, 6))\n    for ch in electrodes_to_plot:\n        if ch in eeg_data.columns:\n            plt.plot(eeg_data[ch].values[:1000], label=ch)  # first 1000 samples\n\n    plt.title(f\"EEG ID: {eeg_id}\")\n    plt.xlabel(\"Time\")\n    plt.ylabel(\"Signal\")\n    plt.legend()\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T13:56:53.381213Z","iopub.execute_input":"2025-05-12T13:56:53.382148Z","iopub.status.idle":"2025-05-12T13:56:55.139861Z","shell.execute_reply.started":"2025-05-12T13:56:53.382116Z","shell.execute_reply":"2025-05-12T13:56:55.138794Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, accuracy_score\nfrom sklearn.preprocessing import LabelEncoder\n\n# We'll use the 'vote' columns as features\nvote_columns = [\"seizure_vote\", \"lpd_vote\", \"gpd_vote\", \"lrda_vote\", \"grda_vote\", \"other_vote\"]\nX = train_df[vote_columns]\n\n# Target column: expert_consensus\nle = LabelEncoder()\ny = le.fit_transform(train_df['expert_consensus'])  # Convert string labels to numbers","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T13:58:49.850717Z","iopub.execute_input":"2025-05-12T13:58:49.851055Z","iopub.status.idle":"2025-05-12T13:58:49.879370Z","shell.execute_reply.started":"2025-05-12T13:58:49.851029Z","shell.execute_reply":"2025-05-12T13:58:49.878380Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split the dataset into training and testing sets (80% train, 20% test)\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Initialize the Random Forest model\nrf_model = RandomForestClassifier(n_estimators=100, random_state=42)\n\n# Train the model\nrf_model.fit(X_train, y_train)\n\n# Make predictions on the test set\ny_pred = rf_model.predict(X_test)\n\n# Evaluate the model\nprint(\"Accuracy:\", accuracy_score(y_test, y_pred))\nprint(\"\\nClassification Report:\\n\", classification_report(y_test, y_pred, target_names=le.classes_))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T14:00:00.072422Z","iopub.execute_input":"2025-05-12T14:00:00.073145Z","iopub.status.idle":"2025-05-12T14:00:03.867286Z","shell.execute_reply.started":"2025-05-12T14:00:00.073118Z","shell.execute_reply":"2025-05-12T14:00:03.866352Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot feature importances\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Get feature importances from the model\nimportances = rf_model.feature_importances_\nfeature_names = X.columns\n\n# Create a DataFrame for plotting\nfeat_imp_df = pd.DataFrame({\n    'Feature': feature_names,\n    'Importance': importances\n}).sort_values(by='Importance', ascending=False)\n\n# Plot\nplt.figure(figsize=(10, 6))\nsns.barplot(x='Importance', y='Feature', data=feat_imp_df, palette='viridis')\nplt.title(\"Random Forest Feature Importances\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T14:00:14.704283Z","iopub.execute_input":"2025-05-12T14:00:14.704646Z","iopub.status.idle":"2025-05-12T14:00:14.924313Z","shell.execute_reply.started":"2025-05-12T14:00:14.704621Z","shell.execute_reply":"2025-05-12T14:00:14.923050Z"}},"outputs":[],"execution_count":null}]}