{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7392775,"sourceType":"datasetVersion","datasetId":4297782}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pyspark.sql import SparkSession\nfrom pyspark.sql.functions import col, udf, explode, array\nfrom pyspark.sql.types import DoubleType, ArrayType, StringType\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom pyspark.ml.feature import VectorAssembler\nfrom pyspark.ml.stat import Correlation\n\n# Initialize Spark Session\nspark = SparkSession.builder \\\n    .appName(\"EEG Analysis\") \\\n    .getOrCreate()\n\n# Path to data\nBASE_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/'\nFILE_PATH = BASE_PATH + 'train_eegs/1000913311.parquet'\n\n\n\n# # Shutdown Spark\n# spark.stop()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T22:24:03.366345Z","iopub.execute_input":"2025-03-15T22:24:03.366781Z","iopub.status.idle":"2025-03-15T22:24:12.563280Z","shell.execute_reply.started":"2025-03-15T22:24:03.366747Z","shell.execute_reply":"2025-03-15T22:24:12.561783Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **About the Problem**\n\nThere are 6 patterns to be identified:\n\nseizure (SZ)\ngeneralized periodic discharges (GPD)\nlateralized periodic discharges (LPD)\nlateralized rhythmic delta activity (LRDA)\ngeneralized rhythmic delta activity (GRDA)\nother\n\nThe annotations were made by a group of experts, however the challenge is that not even the experts can fully agree on a case 100% of the time. Hence, the competition creates a second set of labels:\n\nwhere there are high levels of agreement => “idealized” patterns\nwhere ~1-2 experts give a label as “other” and ~1-2 give one of the remaining five labels => “proto” patterns\nwhere experts are approximately split between 2 of the 5 named patterns => “edge” cases\n","metadata":{}},{"cell_type":"code","source":"df_eeg = spark.read.parquet(FILE_PATH)\ndf_eeg","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T22:24:15.106666Z","iopub.execute_input":"2025-03-15T22:24:15.107447Z","iopub.status.idle":"2025-03-15T22:24:20.667198Z","shell.execute_reply.started":"2025-03-15T22:24:15.107406Z","shell.execute_reply":"2025-03-15T22:24:20.665631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" df_eeg.show(5)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T22:24:20.669128Z","iopub.execute_input":"2025-03-15T22:24:20.669672Z","iopub.status.idle":"2025-03-15T22:24:25.233105Z","shell.execute_reply.started":"2025-03-15T22:24:20.669618Z","shell.execute_reply":"2025-03-15T22:24:25.230892Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" df = spark.read.csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\", \n                     header=True, inferSchema=True)\n\n# Display the first few rows\ndf.show(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T22:24:25.234909Z","iopub.execute_input":"2025-03-15T22:24:25.235315Z","iopub.status.idle":"2025-03-15T22:24:28.232491Z","shell.execute_reply.started":"2025-03-15T22:24:25.235280Z","shell.execute_reply":"2025-03-15T22:24:28.231731Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# if \"eval\" in FLAGS:\n#     import os\n\n#     # Set the environment variable\n#     os.environ[\"PYSPARK_PIN_THREAD\"] = \"False\"\n#     # spark.builder.config(\"spark.jars.packages\", \"org.mlflow.mlflow-spark\")\n#     import mlflow\n\n#     # mlflow.set_tracking_uri(\"http://127.0.0.0:5000\")\n#     mlflow.set_tracking_uri(\"http://localhost:5000\")\n#     mlflow.autolog()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T22:24:28.234934Z","iopub.execute_input":"2025-03-15T22:24:28.235250Z","iopub.status.idle":"2025-03-15T22:24:28.240297Z","shell.execute_reply.started":"2025-03-15T22:24:28.235224Z","shell.execute_reply":"2025-03-15T22:24:28.239449Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" # Extract column names\ncolumns = df.columns\nTARGETS = columns[-6:]\n\n# Print shape (row count and column count)\nprint(\"Train shape:\", (df.count(), len(columns)))\n\n# Display target column names\nprint(\"Target Labels:\", TARGETS)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T22:24:28.241870Z","iopub.execute_input":"2025-03-15T22:24:28.242297Z","iopub.status.idle":"2025-03-15T22:24:29.249356Z","shell.execute_reply.started":"2025-03-15T22:24:28.242253Z","shell.execute_reply":"2025-03-15T22:24:29.247292Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pyspark.sql.functions import col\n\n# Count the number of occurrences of each EEG pattern\nfor target in TARGETS:\n    df.groupBy(target).count().orderBy(col(\"count\").desc()).show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T22:24:29.250354Z","iopub.execute_input":"2025-03-15T22:24:29.250791Z","iopub.status.idle":"2025-03-15T22:24:33.572450Z","shell.execute_reply.started":"2025-03-15T22:24:29.250749Z","shell.execute_reply":"2025-03-15T22:24:33.571209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pyspark.sql.functions import first, min, max, sum, col\n\n# Select the first spectrogram_id and earliest spectrogram_label_offset_seconds for each eeg_id\ntrain = df.groupBy(\"eeg_id\").agg(\n    first(\"spectrogram_id\").alias(\"spec_id\"),\n    min(\"spectrogram_label_offset_seconds\").alias(\"min\")\n)\n\n# Find the latest spectrogram_label_offset_seconds\ntmp = df.groupBy(\"eeg_id\").agg(max(\"spectrogram_label_offset_seconds\").alias(\"max\"))\ntrain = train.join(tmp, on=\"eeg_id\", how=\"left\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T22:24:33.574550Z","iopub.execute_input":"2025-03-15T22:24:33.575219Z","iopub.status.idle":"2025-03-15T22:24:33.704546Z","shell.execute_reply.started":"2025-03-15T22:24:33.575156Z","shell.execute_reply":"2025-03-15T22:24:33.703385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tmp = df.groupBy(\"eeg_id\").agg(first(\"patient_id\").alias(\"patient_id\"))\ntrain = train.join(tmp, on=\"eeg_id\", how=\"left\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T22:24:33.707703Z","iopub.execute_input":"2025-03-15T22:24:33.708134Z","iopub.status.idle":"2025-03-15T22:24:33.766103Z","shell.execute_reply.started":"2025-03-15T22:24:33.708096Z","shell.execute_reply":"2025-03-15T22:24:33.765227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_agg = df.groupBy(\"eeg_id\").agg(\n    *[sum(col(t)).alias(t) for t in TARGETS]  # Sum votes for each target label\n)\n\ntrain = train.join(target_agg, on=\"eeg_id\", how=\"left\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T22:24:33.766972Z","iopub.execute_input":"2025-03-15T22:24:33.767267Z","iopub.status.idle":"2025-03-15T22:24:33.857374Z","shell.execute_reply.started":"2025-03-15T22:24:33.767243Z","shell.execute_reply":"2025-03-15T22:24:33.856180Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T22:24:33.858282Z","iopub.execute_input":"2025-03-15T22:24:33.858686Z","iopub.status.idle":"2025-03-15T22:24:37.230066Z","shell.execute_reply.started":"2025-03-15T22:24:33.858646Z","shell.execute_reply":"2025-03-15T22:24:37.228769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = df\ntrain.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T22:24:37.231051Z","iopub.execute_input":"2025-03-15T22:24:37.231439Z","iopub.status.idle":"2025-03-15T22:24:37.394908Z","shell.execute_reply.started":"2025-03-15T22:24:37.231402Z","shell.execute_reply":"2025-03-15T22:24:37.393509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pyspark.sql.functions import sum as spark_sum\n\nfrom pyspark.sql.functions import monotonically_increasing_id\n\n\nvote_cols = [col for col in train.columns if '_vote' in col] \nprint(\"vote cols:\", vote_cols)\ncolss=[\"eeg_id\", \"spectrogram_id\", \"patient_id\"]\n\n# Group by eeg_id, spectrogram_id, patient_id and sum the vote columns\ntrain_group = train.groupBy(\"eeg_id\", \"spectrogram_id\", \"patient_id\")\\\n                   .agg(*[spark_sum(col).alias(col) for col in vote_cols]).withColumn(\"index_column\", monotonically_increasing_id())\n\ntrain_group.show(7)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T22:24:37.397634Z","iopub.execute_input":"2025-03-15T22:24:37.398118Z","iopub.status.idle":"2025-03-15T22:24:38.838280Z","shell.execute_reply.started":"2025-03-15T22:24:37.398069Z","shell.execute_reply":"2025-03-15T22:24:38.834064Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pyspark.sql.functions import sum as spark_sum, monotonically_increasing_id\n\n# Add an index column if needed (but not necessary for groupBy)\ntrain = train.withColumn(\"index_column\", monotonically_increasing_id())\n\n# Define vote columns\nvote_cols = [col for col in train.columns if '_vote' in col]\nprint(\"Vote cols:\", vote_cols)\n\n# Group by eeg_id, spectrogram_id, patient_id and sum the vote columns\ntrain_group = train.groupBy(\"eeg_id\", \"spectrogram_id\", \"patient_id\")\\\n                   .agg(*[spark_sum(col).alias(col) for col in vote_cols])\n\n# Show the first 7 rows\ntrain_group.show(7)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T22:24:38.839216Z","iopub.execute_input":"2025-03-15T22:24:38.839650Z","iopub.status.idle":"2025-03-15T22:24:39.920357Z","shell.execute_reply.started":"2025-03-15T22:24:38.839607Z","shell.execute_reply":"2025-03-15T22:24:39.919109Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pyspark.sql.functions import col, udf\nfrom pyspark.sql.types import StringType\nimport numpy, pandas\nimport pyspark.pandas as ps\nimport numpy as np\nps.set_option('compute.ops_on_diff_frames', True)\n\ndef categorize_votes(row):\n    # compute max and sum\n    col_names = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n    max_vote = row[col_names].max()\n    total_votes = row[col_names].sum()\n \n    percentage = max_vote / total_votes * 100\n\n    high_agreement_threshold = 70\n    equal_splitting_threshold = 40\n\n    if percentage >= high_agreement_threshold:\n        return 'idealized'\n    elif row['other_vote'] / total_votes >= 0.4 and percentage >= equal_splitting_threshold:\n        return 'proto'\n    elif row['other_vote'] == 0 and percentage >= equal_splitting_threshold:\n        return 'edge'\n    else:\n        return 'undecided'\n\n\n \ntrain_group= ps.DataFrame(train_group)\ntrain_group['pattern'] = train_group.apply(categorize_votes, axis=1)\ntrain_group.head(7)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T22:24:39.924678Z","iopub.execute_input":"2025-03-15T22:24:39.925169Z","iopub.status.idle":"2025-03-15T22:25:01.156902Z","shell.execute_reply.started":"2025-03-15T22:24:39.925127Z","shell.execute_reply":"2025-03-15T22:25:01.155738Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_group[train_group.eeg_id==722738444]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T20:22:03.429535Z","iopub.execute_input":"2025-03-15T20:22:03.430605Z","iopub.status.idle":"2025-03-15T20:22:03.580791Z","shell.execute_reply.started":"2025-03-15T20:22:03.430555Z","shell.execute_reply":"2025-03-15T20:22:03.579587Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Notice the diff between exec times of the above piece of code(15 sec) and below(0.43s) its because above one is a pandas df in spark while below one is pyspark dataframe\n\n","metadata":{}},{"cell_type":"code","source":"train[train.eeg_id==722738444].show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T20:24:02.178640Z","iopub.execute_input":"2025-03-15T20:24:02.179066Z","iopub.status.idle":"2025-03-15T20:24:02.604016Z","shell.execute_reply.started":"2025-03-15T20:24:02.179031Z","shell.execute_reply":"2025-03-15T20:24:02.602915Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":" ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":" ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pyspark.sql import SparkSession\nfrom pyspark.sql.functions import col, mean, min\nimport os\n\n# Initialize Spark session \n\n# Define path to spectrogram parquet files\nPATH = \"/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/\"\n\n# List all parquet files\nfiles = [f for f in os.listdir(PATH) if f.endswith(\".parquet\")]\n\nprint(f\"There are {len(files)} spectrogram parquet files.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T22:25:01.158146Z","iopub.execute_input":"2025-03-15T22:25:01.158788Z","iopub.status.idle":"2025-03-15T22:25:01.248866Z","shell.execute_reply.started":"2025-03-15T22:25:01.158744Z","shell.execute_reply":"2025-03-15T22:25:01.247619Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"spectrogram_id = 789577333\n\n# read in the data\nspec_base_path = \"/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/\"\nspec_data = spark.read.parquet(spec_base_path + str(spectrogram_id) + \".parquet\")\n \nprint((spec_data.count(), len(spec_data.columns)))\nspec_data.show(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T20:29:40.579265Z","iopub.execute_input":"2025-03-15T20:29:40.579708Z","iopub.status.idle":"2025-03-15T20:29:41.702660Z","shell.execute_reply.started":"2025-03-15T20:29:40.579673Z","shell.execute_reply":"2025-03-15T20:29:41.701194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pprint import pprint\n\n# number of spectrograms for each category\nN = 5\n\n# Define the categories\ncategories = [\"seizure_vote\", \"lpd_vote\", \"gpd_vote\", \"lrda_vote\", \"grda_vote\", \"other_vote\"]\n\n# Filter DataFrame to only include \"idealized\" pattern - using correct pandas-on-spark syntax\nidealized_df = train_group[train_group[\"pattern\"] == \"idealized\"].reset_index(drop=True)\n\n# Initialize empty dictionary to store results\nspec_dict = {}\n\n# For each category, find the top N rows by vote value\nfor category in categories:\n    # Sort by the category column and get top N rows\n    top_rows = idealized_df.sort_values(by=category, ascending=False).head(N)\n    \n    # Get the spectrogram_ids as a numpy array\n    spec_dict[category] = top_rows[\"spectrogram_id\"].to_numpy()\n\npprint(spec_dict)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T21:34:20.143640Z","iopub.execute_input":"2025-03-15T21:34:20.144035Z","iopub.status.idle":"2025-03-15T21:35:44.802826Z","shell.execute_reply.started":"2025-03-15T21:34:20.144003Z","shell.execute_reply":"2025-03-15T21:35:44.801656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nsns.set_theme(style=\"darkgrid\")  # Dark theme for contrast\n\ndef plot_spectrograms_by_category(spectrogram_ids, category, cmap=\"bone\"):\n    # Base path for spectrograms\n    spec_base_path = \"/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/\"\n    \n    # Read spectrogram data and handle NaN\n    spec_data = [pd.read_parquet(spec_base_path + str(id) + \".parquet\").fillna(0) for id in spectrogram_ids]  \n\n    # Avoid log(0) issues\n    epsilon = 1e-10  \n\n    # 🔥 Make the plots **bigger**\n    fig, axes = plt.subplots(1, len(spectrogram_ids), figsize=(30, 10), sharey=False)  # Bigger size\n    plt.suptitle(f\"{category}\", fontsize=24, fontweight=\"bold\", color=\"black\")\n\n    if len(spectrogram_ids) == 1:\n        axes = [axes]  # Ensure it's always iterable\n\n    for i, ax in enumerate(axes):\n        # Apply log transformation safely\n        log_spec_data = np.log(np.maximum(spec_data[i].T, epsilon))  # Avoid NaN & log(0)\n        \n        # Plot spectrogram with **bigger size**\n        img = ax.imshow(log_spec_data, cmap=cmap, aspect=\"auto\")\n\n        # Set labels and title\n        ax.set_title(f'ID {spectrogram_ids[i]}', fontsize=18, fontweight=\"bold\", color=\"black\")\n        ax.set_xlabel(\"Time\", fontsize=18, fontweight=\"bold\", color=\"black\")\n        ax.set_ylabel(\"Frequency (Hz)\", fontsize=18, fontweight=\"bold\", color=\"black\")\n\n        # Set ticks color\n        ax.tick_params(axis='both', labelsize=14, colors=\"black\")\n        axes[i].tick_params(axis='both', which='both', labelsize=10)\n\n        # Add color bar for each spectrogram\n        cbar = fig.colorbar(img, ax=ax, fraction=0.02, pad=0.04)\n        cbar.ax.tick_params(labelsize=14, colors=\"black\")\n\n    plt.tight_layout()  # Ensure labels are not cropped\n    plt.subplots_adjust(top=0.85)\n    plt.show()\n\n# Plot spectrograms for each category using high-contrast color maps\ndark_color_maps = [\"bone\", \"gray\", \"afmhot\", \"gist_heat\", \"cividis\", \"inferno\"]\nfor i, (key, values) in enumerate(spec_dict.items()):\n    plot_spectrograms_by_category(spectrogram_ids=values, category=key, cmap=dark_color_maps[i % len(dark_color_maps)])\n ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T21:51:16.737607Z","iopub.execute_input":"2025-03-15T21:51:16.737964Z","iopub.status.idle":"2025-03-15T21:51:37.330202Z","shell.execute_reply.started":"2025-03-15T21:51:16.737932Z","shell.execute_reply":"2025-03-15T21:51:37.328873Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" \nimport os\nimport gc\nimport wandb\nimport random\nimport math\nfrom glob import glob\nfrom tqdm import tqdm\nfrom time import time\nfrom pprint import pprint\nimport warnings\nimport pandas as pd\nimport numpy as np\nfrom scipy.signal import spectrogram\n\n# visuals\nimport seaborn as sns\nimport matplotlib as mpl\nfrom matplotlib import cm\nimport matplotlib.patches as patches\nimport matplotlib.pyplot as plt\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T22:29:40.630699Z","iopub.execute_input":"2025-03-15T22:29:40.631590Z","iopub.status.idle":"2025-03-15T22:29:46.764144Z","shell.execute_reply.started":"2025-03-15T22:29:40.631521Z","shell.execute_reply":"2025-03-15T22:29:46.762864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" spect_data = np.load(\"/kaggle/input/brain-spectrograms/specs.npy\", allow_pickle=True).item()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T22:27:16.801029Z","iopub.execute_input":"2025-03-15T22:27:16.801415Z","iopub.status.idle":"2025-03-15T22:27:29.043983Z","shell.execute_reply.started":"2025-03-15T22:27:16.801385Z","shell.execute_reply":"2025-03-15T22:27:29.042675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nprint(spect_data[319287046])\nprint(spect_data[319287046].shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T22:29:46.765868Z","iopub.execute_input":"2025-03-15T22:29:46.766641Z","iopub.status.idle":"2025-03-15T22:29:46.774614Z","shell.execute_reply.started":"2025-03-15T22:29:46.766600Z","shell.execute_reply":"2025-03-15T22:29:46.772858Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n# Step 3: Extract the NumPy array from the dictionary\nspectrogram_array = list(spect_data.values())[0]  # Extract the first value (NumPy array) from the dictionary\n\n# Step 4: Load column names from the sample Parquet file using PySpark\nsample_path = \"/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/1000086677.parquet\"\nfeature_col_names = spark.read.parquet(sample_path).columns[1:]  # Skip the first column (e.g., 'time' or 'index')\n\n# Step 5: Convert the NumPy array to a Pandas DataFrame with the extracted column names\npdf = pd.DataFrame(spectrogram_array, columns=feature_col_names)\n\n# Step 6: Convert the Pandas DataFrame to a PySpark DataFrame\nppdf = spark.createDataFrame(pdf)\n\n# Step 7: Show the DataFrame\n# df.show(1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T22:29:59.993654Z","iopub.execute_input":"2025-03-15T22:29:59.994090Z","iopub.status.idle":"2025-03-15T22:30:03.237172Z","shell.execute_reply.started":"2025-03-15T22:29:59.994055Z","shell.execute_reply":"2025-03-15T22:30:03.236242Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nfe_data = {}\n\n# Iterate over spectrogram data with tqdm for progress tracking\nfor spect_id, data in tqdm(spect_data.items()):  # ✅ Correct usage\n    fe_data[spect_id] = {}\n    \n    for k, feature in enumerate(feature_col_names):\n        fe_data[spect_id][f\"{feature}_mean\"] = data[:, k].mean()\n        fe_data[spect_id][f\"{feature}_min\"] = data[:, k].min()\n        fe_data[spect_id][f\"{feature}_max\"] = data[:, k].max()\n        fe_data[spect_id][f\"{feature}_std\"] = data[:, k].std()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T23:10:49.316200Z","iopub.execute_input":"2025-03-15T23:10:49.316692Z","iopub.status.idle":"2025-03-15T23:14:19.607243Z","shell.execute_reply.started":"2025-03-15T23:10:49.316648Z","shell.execute_reply":"2025-03-15T23:14:19.606018Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pyspark.sql import SparkSession\nfrom pyspark.sql.functions import col, first\nfrom pyspark.ml.feature import StringIndexer\nimport pandas as pd\n# Step 2: Convert fe_data dictionary to a PySpark DataFrame\nfe_data_df = spark.createDataFrame(pd.DataFrame.from_dict(fe_data, orient='index').reset_index())\n\n# Step 3: Load target labels from the train DataFrame\n# Assuming `train` is already a PySpark DataFrame\n# If not, convert it to a PySpark DataFrame:\n# train = spark.createDataFrame(train)\n\n# Group by spectrogram_id and get the first expert_consensus value\ntarget_df = train.groupBy(\"spectrogram_id\").agg(first(\"expert_consensus\").alias(\"expert_consensus\"))\n\n# Step 4: Encode expert_consensus from string to numerical values\n# Use StringIndexer to encode the expert_consensus column\nstring_indexer = StringIndexer(inputCol=\"expert_consensus\", outputCol=\"expert_consensus_encoded\")\ntarget_df = string_indexer.fit(target_df).transform(target_df)\n\n# Step 5: Merge fe_data_df with target_df on spectrogram_id\nfinal_df = fe_data_df.join(target_df, fe_data_df[\"index\"] == target_df[\"spectrogram_id\"], how=\"inner\")\n\n# # Step 6: Show the final DataFrame\n# final_df.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T23:31:25.052844Z","iopub.execute_input":"2025-03-15T23:31:25.053344Z","iopub.status.idle":"2025-03-15T23:36:34.659401Z","shell.execute_reply.started":"2025-03-15T23:31:25.053309Z","shell.execute_reply":"2025-03-15T23:36:34.658067Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_df = final_df.drop(\"expert_consensus\")\nprint(final_df.count(), len(final_df.columns))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T23:58:32.240154Z","iopub.execute_input":"2025-03-15T23:58:32.240674Z","iopub.status.idle":"2025-03-15T23:58:36.457090Z","shell.execute_reply.started":"2025-03-15T23:58:32.240635Z","shell.execute_reply":"2025-03-15T23:58:36.455611Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Step 2: Display the first 10 and last 10 column names\ncol_names = final_df.columns\n\nprint(\"First 10 column names:\")\nprint(col_names[:10])\n\nprint(\"\\nLast 10 column names:\")\nprint(col_names[-10:])\n\n# # Step 3: Display 'expert_consensus' and 'expert_consensus_encoded' for the first 5 rows\n# final_df.select(\"expert_consensus\", \"expert_consensus_encoded\").show(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T23:58:46.549263Z","iopub.execute_input":"2025-03-15T23:58:46.549670Z","iopub.status.idle":"2025-03-15T23:58:46.557725Z","shell.execute_reply.started":"2025-03-15T23:58:46.549638Z","shell.execute_reply":"2025-03-15T23:58:46.556597Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Select distinct pairs of expert_consensus and expert_consensus_encoded\nmapping_df = final_df.select(\"expert_consensus\", \"expert_consensus_encoded\").distinct()\n\n# Sort by expert_consensus_encoded for better readability\nmapping_df = mapping_df.orderBy(\"expert_consensus_encoded\")\n\n# Show the mapping\nmapping_df.show(truncate=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T23:51:21.761840Z","iopub.execute_input":"2025-03-15T23:51:21.762277Z","iopub.status.idle":"2025-03-15T23:51:25.967398Z","shell.execute_reply.started":"2025-03-15T23:51:21.762244Z","shell.execute_reply":"2025-03-15T23:51:25.965901Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"index_value = 319287046\nfiltered_df = final_df.filter(col(\"index\") == index_value)\n\n# Step 3: Select and display the required columns\nresult = filtered_df.select(\"index\", \"spectrogram_id\", \"expert_consensus\", \"expert_consensus_encoded\")\n\n# Step 4: Show the result\nresult.show(truncate=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T23:48:03.724348Z","iopub.execute_input":"2025-03-15T23:48:03.724973Z","iopub.status.idle":"2025-03-15T23:48:08.292397Z","shell.execute_reply.started":"2025-03-15T23:48:03.724926Z","shell.execute_reply":"2025-03-15T23:48:08.290875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from bigdl.dllib.nncontext import init_nncontext\nfrom bigdl.dllib.feature.image import *\nfrom bigdl.dllib.keras.models import Model\nfrom bigdl.dllib.keras.layers import Input, Conv2D, MaxPooling2D, Flatten, Dense, LSTM, Reshape\nfrom bigdl.dllib.optim.optimizer import Adam, MaxIteration, ClassNLLCriterion\nfrom pyspark.sql import SparkSession\nfrom bigdl.dllib.utils.common import Sample\n\n# # 🚀 Initialize Spark + BigDL Context\n# spark = SparkSession.builder.appName(\"EEG_Classification\").getOrCreate()\nsc = init_nncontext()\n\n# 🔹 Convert Spark DataFrame to BigDL FeatureSet\ndef df_to_featureset(df, feature_cols, label_col):\n    rdd = df.rdd.map(lambda row: \n        Sample.from_ndarray(\n            row[feature_cols], \n            row[label_col]\n        )\n    )\n    return rdd\n\n# 🚀 Prepare Data\n# Exclude index, expert_consensus, and expert_consensus_encoded columns\nfeature_cols = [col for col in final_df.columns if col not in [\"index\", \"expert_consensus\", \"expert_consensus_encoded\"]]\nlabel_col = \"expert_consensus_encoded\"  # Use the encoded label column\n\ntrain_rdd = df_to_featureset(final_df, feature_cols, label_col)\n\n# 🔥 Define CNN + LSTM Model\ninput_layer = Input(shape=(len(feature_cols),))  # EEG features\nx = Reshape((1, len(feature_cols), 1))(input_layer)  # Reshape for CNN\n\nx = Conv2D(32, kernel_size=(1,3), activation=\"relu\", padding=\"same\")(x)\nx = MaxPooling2D(pool_size=(1,2))(x)\nx = Flatten()(x)\n\nx = Reshape((1, -1))(x)  # Reshape for LSTM\nx = LSTM(64, return_sequences=False)(x)\noutput_layer = Dense(3, activation=\"softmax\")(x)  # 3-class classification (adjust based on your data)\n\nmodel = Model(input_layer, output_layer)\n\n# 🔹 Compile & Train Model\noptimizer = Adam(learningrate=0.001)\nmodel.compile(optimizer=optimizer, loss=ClassNLLCriterion())\n\n# 🔥 Train the Model\nmodel.fit(train_rdd, batch_size=32, nb_epoch=10)\n\n# 🎯 Save & Evaluate\nmodel.save(\"/tmp/eeg_cnn_lstm.bigdl\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Train shape:\", (train.count(), len(train.columns)), \"\\n\")  # PySpark equivalent of shape\nprint(\"Unique eeg_ids: \", train.select(\"eeg_id\").distinct().count())\nprint(train.groupBy(\"eeg_id\").count().describe().show(), \"\\n\")\nprint(\"Unique spectrogram_ids: \", train.select(\"spec_id\").distinct().count())\nprint(\"Unique patient_ids: \", train.select(\"patient_id\").distinct().count(), \"\\n\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T22:04:56.534750Z","iopub.execute_input":"2025-03-15T22:04:56.535134Z","iopub.status.idle":"2025-03-15T22:04:59.612351Z","shell.execute_reply.started":"2025-03-15T22:04:56.535101Z","shell.execute_reply":"2025-03-15T22:04:59.610916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!python --version\n!pip show pyspark\n!pip show bigdl-dllib\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T00:23:24.862973Z","iopub.execute_input":"2025-03-16T00:23:24.863308Z","iopub.status.idle":"2025-03-16T00:23:30.593280Z","shell.execute_reply.started":"2025-03-16T00:23:24.863277Z","shell.execute_reply":"2025-03-16T00:23:30.592418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install pyspark==3.3.0 --force-reinstall","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T00:24:18.516611Z","iopub.execute_input":"2025-03-16T00:24:18.516940Z","iopub.status.idle":"2025-03-16T00:24:49.566709Z","shell.execute_reply.started":"2025-03-16T00:24:18.516912Z","shell.execute_reply":"2025-03-16T00:24:49.565907Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Let's check what modules are available in bigdl.dllib.keras\nimport bigdl.dllib.keras as keras\nprint(dir(keras))\n\n# Let's also check what's in the bigdl.dllib package\nimport bigdl.dllib as dllib\nprint(dir(dllib))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T00:36:14.108653Z","iopub.execute_input":"2025-03-16T00:36:14.108996Z","iopub.status.idle":"2025-03-16T00:36:14.114289Z","shell.execute_reply.started":"2025-03-16T00:36:14.108967Z","shell.execute_reply":"2025-03-16T00:36:14.113270Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install bigdl-spark3  # Install the latest release version","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T00:31:24.394483Z","iopub.execute_input":"2025-03-16T00:31:24.394773Z","iopub.status.idle":"2025-03-16T00:31:57.850803Z","shell.execute_reply.started":"2025-03-16T00:31:24.394746Z","shell.execute_reply":"2025-03-16T00:31:57.849912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from bigdl.dllib.nncontext import init_nncontext\nfrom bigdl.dllib.feature.image import *\nfrom bigdl.dllib.keras.models import Model\nfrom bigdl.dllib.keras.layers import Input, Conv2D, MaxPooling2D, Flatten, Dense, LSTM, Reshape\nfrom bigdl.dllib.optim.optimizer import Adam, MaxIteration\nfrom bigdl.dllib.keras.objectives import CategoricalCrossEntropy # Use this instead\nfrom pyspark.sql import SparkSession\nfrom bigdl.dllib.utils.common import Sample","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T00:40:59.082967Z","iopub.execute_input":"2025-03-16T00:40:59.083327Z","iopub.status.idle":"2025-03-16T00:40:59.087897Z","shell.execute_reply.started":"2025-03-16T00:40:59.083300Z","shell.execute_reply":"2025-03-16T00:40:59.086937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" \nsc = init_nncontext()\n\n# 🔹 Convert Spark DataFrame to BigDL FeatureSet\ndef df_to_featureset(df, feature_cols, label_col):\n    rdd = df.rdd.map(lambda row: \n        Sample.from_ndarray(\n            row[feature_cols], \n            row[label_col]\n        )\n    )\n    return rdd\n\n# 🚀 Prepare Data\n# Exclude index, expert_consensus, and expert_consensus_encoded columns\nfeature_cols = [col for col in final_df.columns if col not in [\"index\",   \"expert_consensus_encoded\"]]\nlabel_col = \"expert_consensus_encoded\"  # Use the encoded label column\n\ntrain_rdd = df_to_featureset(final_df, feature_cols, label_col)\n\n# 🔥 Define CNN + LSTM Model\ninput_layer = Input(shape=(len(feature_cols),))  # EEG features\nx = Reshape((1, len(feature_cols), 1))(input_layer)  # Reshape for CNN\n\nx = Conv2D(32, kernel_size=(1,3), activation=\"relu\", padding=\"same\")(x)\nx = MaxPooling2D(pool_size=(1,2))(x)\nx = Flatten()(x)\n\nx = Reshape((1, -1))(x)  # Reshape for LSTM\nx = LSTM(64, return_sequences=False)(x)\noutput_layer = Dense(3, activation=\"softmax\")(x)  # 3-class classification (adjust based on your data)\n\nmodel = Model(input_layer, output_layer)\n\n# 🔹 Compile & Train Model\noptimizer = Adam(learningrate=0.001)\nmodel.compile(optimizer=optimizer, loss=ClassNLLCriterion())\n\n# 🔥 Train the Model\nmodel.fit(train_rdd, batch_size=32, nb_epoch=10)\n\n# 🎯 Save & Evaluate\nmodel.save(\"/kaggle/working/eeg_cnn_lstm.bigdl\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T00:41:50.752673Z","iopub.execute_input":"2025-03-16T00:41:50.753025Z","iopub.status.idle":"2025-03-16T00:41:50.772158Z","shell.execute_reply.started":"2025-03-16T00:41:50.752995Z","shell.execute_reply":"2025-03-16T00:41:50.770982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n!pip install bigdl --upgrade","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T00:20:08.653005Z","iopub.execute_input":"2025-03-16T00:20:08.653344Z","iopub.status.idle":"2025-03-16T00:21:15.866066Z","shell.execute_reply.started":"2025-03-16T00:20:08.653317Z","shell.execute_reply":"2025-03-16T00:21:15.865259Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"spectrogram_id = 789577333\n\n# read in the data\nspec_base_path = \"/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/\"\nspec_data = pd.read_parquet(spec_base_path + str(spectrogram_id) + \".parquet\")\n\nprint(spec_data.shape)\n\nspec_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T10:05:30.125519Z","iopub.execute_input":"2025-03-15T10:05:30.125943Z","iopub.status.idle":"2025-03-15T10:05:30.399185Z","shell.execute_reply.started":"2025-03-15T10:05:30.125917Z","shell.execute_reply":"2025-03-15T10:05:30.398084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(\"Frequency Columns:\", freq_cols)\n# print(\"Available Columns:\", df.columns)\n# # Identify spectrogram feature columns (exclude non-numeric and timestamp)\n# freq_cols = [c for c in df.columns if c.startswith((\"LL_\", \"RL_\", \"LP_\", \"RP_\"))]\n\n# # Debug: Print extracted frequency columns\n# print(\"✅ Frequency Columns:\", freq_cols)\n\n# # If still empty, raise an error\n# if not freq_cols:\n#     raise ValueError(\"❌ No frequency columns found! Check column naming pattern.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-13T16:57:51.755754Z","iopub.execute_input":"2025-03-13T16:57:51.756086Z","iopub.status.idle":"2025-03-13T16:57:51.759845Z","shell.execute_reply.started":"2025-03-13T16:57:51.756061Z","shell.execute_reply":"2025-03-13T16:57:51.758934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"edge_df = train_group[train_group.pattern==\"edge\"] \n \nedge_df.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T10:23:29.666498Z","iopub.execute_input":"2025-03-15T10:23:29.666835Z","iopub.status.idle":"2025-03-15T10:23:31.438515Z","shell.execute_reply.started":"2025-03-15T10:23:29.666809Z","shell.execute_reply":"2025-03-15T10:23:31.436385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T12:02:46.136884Z","iopub.execute_input":"2025-03-15T12:02:46.137111Z","iopub.status.idle":"2025-03-15T12:02:46.231458Z","shell.execute_reply.started":"2025-03-15T12:02:46.137087Z","shell.execute_reply":"2025-03-15T12:02:46.230011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}