{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":10972181,"sourceType":"datasetVersion","datasetId":6827388}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\npath = '/kaggle/input/coefficients/'\nprint(os.listdir(path))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T07:05:42.523496Z","iopub.execute_input":"2025-03-10T07:05:42.523733Z","iopub.status.idle":"2025-03-10T07:05:42.532228Z","shell.execute_reply.started":"2025-03-10T07:05:42.523706Z","shell.execute_reply":"2025-03-10T07:05:42.531407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import numpy as np\n# signal = np.load('1000913311_0_coeff.npy', allow_pickle=True)\n# print(len(signal))\nimport numpy as np\n\nfile_path = '/kaggle/input/coefficients/coefficients/1001487592_2_coeff.npy'\ndata = np.load(file_path, allow_pickle=True).item()\n\n# Check available keys\nprint(data.keys())\n# data = np.load(file_path, allow_pickle=True).item()\n# print(data)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\nfile_path = '/kaggle/input/coefficients/coefficients/1000913311_0_coeff.npy'\n\n# Load the file\nsignal = np.load(file_path, allow_pickle=True)\n\n# Unpack if it's a scalar\nif signal.ndim == 0:\n    signal = signal.item()\n\n# Check the length if it's a dictionary or iterable\nif hasattr(signal, '__len__'):\n    print(f\"Length of signal: {len(signal)}\")\nelse:\n    print(f\"Unsupported type: {type(signal)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T07:11:27.251577Z","iopub.execute_input":"2025-03-10T07:11:27.251842Z","iopub.status.idle":"2025-03-10T07:11:27.261381Z","shell.execute_reply.started":"2025-03-10T07:11:27.251822Z","shell.execute_reply":"2025-03-10T07:11:27.260513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\n\ndata_path = '/kaggle/input/coefficients/coefficients/'\n\n# Get list of .npy files\nnpy_files = [f for f in os.listdir(data_path) if f.endswith('.npy')]\nprint(f\"Found {len(npy_files)} .npy files.\")\n\nsignal_lengths = []\n\n# Load and measure signals\nfor file in npy_files:\n    try:\n        file_path = os.path.join(data_path, file)\n        signal = np.load(file_path, allow_pickle=True)\n\n        # Unpack if it's a scalar\n        if signal.ndim == 0:\n            signal = signal.item()\n\n        # Get length if possible\n        if hasattr(signal, '__len__'):\n            signal_lengths.append((file, len(signal)))\n        else:\n            print(f\"Skipping file {file} (unsupported type: {type(signal)})\")\n\n    except Exception as e:\n        print(f\"Error loading file {file}: {e}\")\n\n# Create a DataFrame for analysis\ndf = pd.DataFrame(signal_lengths, columns=['file', 'length'])\n\n# Check if all signals have the same length\nif df['length'].nunique() == 1:\n    print(f\"✅ All signals have the same length: {df['length'].iloc[0]}\")\nelse:\n    print(\"❌ Signals have different lengths.\")\n\n    # Get stats\n    min_length = df['length'].min()\n    max_length = df['length'].max()\n    median_length = df['length'].median()\n\n    lower_count = df[df['length'] < median_length].shape[0]\n    higher_count = df[df['length'] > median_length].shape[0]\n    medium_count = df[df['length'] == median_length].shape[0]\n\n    print(f\"\\n📏 Min length: {min_length}\")\n    print(f\"📏 Max length: {max_length}\")\n    print(f\"📏 Median length: {median_length}\")\n    print(f\"🔽 Lower than median: {lower_count}\")\n    print(f\"🔼 Higher than median: {higher_count}\")\n    print(f\"⚖️ Same as median: {medium_count}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T07:25:34.137662Z","iopub.execute_input":"2025-03-10T07:25:34.138015Z","iopub.status.idle":"2025-03-10T07:26:42.411358Z","shell.execute_reply.started":"2025-03-10T07:25:34.137991Z","shell.execute_reply":"2025-03-10T07:26:42.410739Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\n\ndata_path = '/kaggle/input/coefficients/coefficients/'\n\n# Get list of .npy files\nnpy_files = [f for f in os.listdir(data_path) if f.endswith('.npy')]\nprint(f\"Found {len(npy_files)} .npy files.\")\n\nsignal_lengths = []\n\n# Load and measure signals\nfor file in npy_files:\n    try:\n        file_path = os.path.join(data_path, file)\n        signal = np.load(file_path, allow_pickle=True)\n\n        # Unpack if it's a scalar\n        if signal.ndim == 0:\n            signal = signal.item()\n\n        # Get length if possible\n        if hasattr(signal, '__len__'):\n            signal_lengths.append({'file': file, 'length': len(signal)})\n        else:\n            signal_lengths.append({'file': file, 'length': None})\n\n    except Exception as e:\n        signal_lengths.append({'file': file, 'length': f\"Error: {e}\"})\n\n# Convert to DataFrame\ndf = pd.DataFrame(signal_lengths)\n\n# Display the result\nprint(df.head(10))  # Display first 10 entries\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T07:17:52.227054Z","iopub.execute_input":"2025-03-10T07:17:52.227352Z","iopub.status.idle":"2025-03-10T07:22:51.718295Z","shell.execute_reply.started":"2025-03-10T07:17:52.227328Z","shell.execute_reply":"2025-03-10T07:22:51.717375Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom concurrent.futures import ThreadPoolExecutor\n\ndata_path = '/kaggle/input/coefficients/coefficients/'\n\n# Get list of .npy files\nnpy_files = [f for f in os.listdir(data_path) if f.endswith('.npy')]\nprint(f\"Found {len(npy_files)} .npy files.\")\n\nsignal_lengths = []\n\n# Function to load and measure signal length\ndef get_signal_length(file):\n    try:\n        file_path = os.path.join(data_path, file)\n        signal = np.load(file_path, allow_pickle=True)\n\n        # Unpack if it's a scalar\n        if signal.ndim == 0:\n            signal = signal.item()\n\n        # Get length if possible\n        if hasattr(signal, '__len__'):\n            return {'file': file, 'length': len(signal)}\n        else:\n            return {'file': file, 'length': None}\n\n    except Exception as e:\n        return {'file': file, 'length': f\"Error: {e}\"}\n\n# Use ThreadPoolExecutor for parallel processing\nwith ThreadPoolExecutor(max_workers=8) as executor:\n    signal_lengths = list(executor.map(get_signal_length, npy_files))\n\n# Convert to DataFrame\ndf = pd.DataFrame(signal_lengths)\n\n# Display first 10 results\nprint(df.head(10))\n\n# Save to CSV (Optional)\ndf.to_csv('/kaggle/working/signal_lengths.csv', index=False)\nprint(\"\\n✅ Results saved to 'signal_lengths.csv'\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T07:23:34.95722Z","iopub.execute_input":"2025-03-10T07:23:34.957572Z","iopub.status.idle":"2025-03-10T07:24:07.691254Z","shell.execute_reply.started":"2025-03-10T07:23:34.957542Z","shell.execute_reply":"2025-03-10T07:24:07.690312Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\ndata_path = '/kaggle/input/coefficients/coefficients/'\n\n# Get list of .npy files\nnpy_files = [f for f in os.listdir(data_path) if f.endswith('.npy')]\nprint(f\"Found {len(npy_files)} .npy files.\")\n\nlabels = []\n\n# Function to extract label\ndef get_label(file):\n    try:\n        file_path = os.path.join(data_path, file)\n        signal = np.load(file_path, allow_pickle=True)\n\n        # Unpack if it's a scalar\n        if signal.ndim == 0:\n            signal = signal.item()\n\n        # Get label if it exists in the dictionary\n        if isinstance(signal, dict) and 'expert_consensus' in signal:\n            labels.append(signal['expert_consensus'])\n        else:\n            labels.append('Unknown')\n\n    except Exception as e:\n        labels.append(f\"Error: {e}\")\n\n# Process files\nfor file in npy_files:\n    get_label(file)\n\n# Create a DataFrame\ndf = pd.DataFrame(labels, columns=['label'])\n\n# Count occurrences of each label\nlabel_counts = df['label'].value_counts()\n\n# Plot the pie chart\nplt.figure(figsize=(8, 8))\nplt.pie(label_counts, labels=label_counts.index, autopct='%1.1f%%', startangle=140, colors=plt.cm.Paired(range(len(label_counts))))\nplt.title('Distribution of Expert Consensus Labels')\nplt.axis('equal')  # Equal aspect ratio ensures the pie chart is a circle\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T08:55:11.890583Z","iopub.execute_input":"2025-03-10T08:55:11.890933Z","iopub.status.idle":"2025-03-10T09:02:21.858724Z","shell.execute_reply.started":"2025-03-10T08:55:11.890903Z","shell.execute_reply":"2025-03-10T09:02:21.857288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\ndata_path = '/kaggle/input/coefficients/coefficients/'\n\n# Get list of .npy files\nnpy_files = [f for f in os.listdir(data_path) if f.endswith('.npy')]\nprint(f\"Found {len(npy_files)} .npy files.\")\n\n# Dictionary to store label counts\nlabel_counts = {\n    'Seizure': 0,\n    'LRDA': 0,\n    'GRDA': 0,\n    'LPD': 0,\n    'GPD': 0,\n    'Others': 0\n}\n\n# Function to extract labels\ndef get_label(file):\n    try:\n        file_path = os.path.join(data_path, file)\n        data = np.load(file_path, allow_pickle=True).item()\n        label = data.get('expert consensus', 'Others')\n\n        if label in label_counts:\n            label_counts[label] += 1\n        else:\n            label_counts['Others'] += 1\n    except Exception as e:\n        print(f\"Error loading {file}: {e}\")\n\n# Process all files\nfor file in npy_files:\n    get_label(file)\n\n# Convert to DataFrame\ndf = pd.DataFrame.from_dict(label_counts, orient='index', columns=['count'])\ndf['percentage'] = (df['count'] / df['count'].sum()) * 100\n\n# Plot pie chart\nplt.figure(figsize=(8, 8))\nplt.pie(df['percentage'], labels=df.index, autopct='%1.1f%%', startangle=140, colors=plt.cm.tab10.colors)\nplt.title('Distribution of Expert Consensus Labels')\nplt.show()\n\n# Display the result\nprint(df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T10:45:39.059332Z","iopub.execute_input":"2025-03-10T10:45:39.059765Z","iopub.status.idle":"2025-03-10T10:52:50.570055Z","shell.execute_reply.started":"2025-03-10T10:45:39.059724Z","shell.execute_reply":"2025-03-10T10:52:50.568822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# File path\nfile_path = '/kaggle/input/hms-harmful-brain-activity-classification/train.csv'\n\n# Load the data\ndata = pd.read_csv(file_path)\n\n# Check for the 'exposures' column\nif 'exposures' in data.columns:\n    # Count the frequency of each category\n    exposure_counts = data['exposures'].value_counts()\n    \n    # Calculate the percentage of each category\n    exposure_percentage = (exposure_counts / exposure_counts.sum()) * 100\n    \n    print(\"Exposure Categories and Percentages:\")\n    print(exposure_percentage)\nelse:\n    print(\"'exposures' column not found in the dataset.\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\n\n# File paths\ncoefficients_file = '/kaggle/input/hms-harmful-brain-activity-classification/coefficients.csv'\ntrain_file = '/kaggle/input/hms-harmful-brain-activity-classification/train.csv'\n\n# Load datasets\ncoefficients = pd.read_csv(coefficients_file)\ntrain = pd.read_csv(train_file)\n\n# Merge both datasets based on a common key (like 'file')\n# Change 'file' to the actual column used for matching\nmerged = pd.merge(coefficients, train, on='file', how='inner')\n\n# Define target labels\ntarget_labels = ['Seizure', 'LRDA', 'GRDA', 'LPD', 'GPD']\n\n# Clean up expert_consensus column\nmerged['expert_consensus_cleaned'] = merged['expert_consensus'].apply(\n    lambda x: x if x in target_labels else 'Others'\n)\n\n# Count the occurrences of each category\ncategory_counts = merged['expert_consensus_cleaned'].value_counts()\n\n# Plot the pie chart\nplt.figure(figsize=(8, 8))\nplt.pie(category_counts, labels=category_counts.index, autopct='%.2f%%', startangle=140, colors=plt.cm.tab10.colors)\nplt.title('Expert Consensus Distribution (Aligned Data)')\nplt.axis('equal')  # Ensures the pie chart is a circle\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}