{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":31259,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# HMS – I/O Anatomy (no training, no scoring)\n\nfrom pathlib import Path\nimport os\nimport json\nimport numpy as np\nimport pandas as pd\n\nDATA_DIR = Path('/kaggle/input')\n\ndef human_size(n: int) -> str:\n    for unit in ['B', 'KB', 'MB', 'GB', 'TB']:\n        if n < 1024:\n            return f'{n:.1f}{unit}'\n        n /= 1024\n    return f'{n:.1f}PB'\n\ndef list_tree(root: Path, max_items: int = 200):\n    print('=== /kaggle/input tree (top) ===')\n    items = []\n    for p in root.rglob('*'):\n        if p.is_file():\n            try:\n                size = p.stat().st_size\n            except Exception:\n                size = -1\n            items.append((str(p), size))\n    items.sort(key=lambda x: x[1], reverse=True)\n    for i, (path, size) in enumerate(items[:max_items], 1):\n        print(f'{i:03d}  {human_size(size)}  {path}')\n    print(f'... total files: {len(items)}')\n\ndef peek_csv(path: Path, n: int = 3):\n    print(f'\\n--- CSV peek: {path} ---')\n    df = pd.read_csv(path)\n    print('shape:', df.shape)\n    print('columns:', list(df.columns))\n    print(df.head(n))\n    return df\n\ndef peek_parquet(path: Path, n: int = 3):\n    print(f'\\n--- Parquet peek: {path} ---')\n    df = pd.read_parquet(path)\n    print('shape:', df.shape)\n    print('columns:', list(df.columns))\n    print(df.head(n))\n    return df\n\ndef peek_jsonl(path: Path, n: int = 3):\n    print(f'\\n--- JSONL peek: {path} ---')\n    rows = []\n    with path.open('r', encoding='utf-8') as f:\n        for i, line in enumerate(f):\n            if i >= n:\n                break\n            rows.append(json.loads(line))\n    print('n_peek:', len(rows))\n    if rows:\n        print('keys:', list(rows[0].keys()))\n        print('row0:', rows[0])\n    return rows\n\ndef peek_npz(path: Path):\n    print(f'\\n--- NPZ peek: {path} ---')\n    z = np.load(path, allow_pickle=True)\n    keys = list(z.keys())\n    print('keys:', keys)\n    for k in keys[:10]:\n        arr = z[k]\n        print(f'  {k}: type={type(arr)} shape={getattr(arr, \"shape\", None)} dtype={getattr(arr, \"dtype\", None)}')\n    return z\n\ndef peek_npy(path: Path):\n    print(f'\\n--- NPY peek: {path} ---')\n    arr = np.load(path, allow_pickle=True)\n    print('type:', type(arr))\n    print('shape:', getattr(arr, 'shape', None))\n    print('dtype:', getattr(arr, 'dtype', None))\n    return arr\n\ndef peek_hdf5(path: Path):\n    print(f'\\n--- HDF5 peek: {path} ---')\n    import h5py\n    with h5py.File(path, 'r') as f:\n        def walk(name, obj):\n            if hasattr(obj, 'shape'):\n                print(f'  {name}: shape={obj.shape} dtype={obj.dtype}')\n            else:\n                print(f'  {name}: group')\n        f.visititems(walk)\n\ndef choose_and_peek(paths):\n    # 優先的に見たい候補: sample submission / metadata / train/test index\n    preferred = []\n    for p in paths:\n        name = p.name.lower()\n        if 'sample' in name and ('submission' in name or 'sub' in name):\n            preferred.append(p)\n        elif 'submission' in name:\n            preferred.append(p)\n        elif 'train' in name or 'test' in name or 'metadata' in name:\n            preferred.append(p)\n    # まず優先を、なければ上から\n    targets = preferred[:10] if preferred else paths[:10]\n\n    for p in targets:\n        suf = p.suffix.lower()\n        try:\n            if suf == '.csv':\n                peek_csv(p, n=5)\n            elif suf == '.parquet':\n                peek_parquet(p, n=5)\n            elif suf in ('.jsonl', '.ndjson'):\n                peek_jsonl(p, n=3)\n            elif suf == '.npz':\n                peek_npz(p)\n            elif suf == '.npy':\n                peek_npy(p)\n            elif suf in ('.h5', '.hdf5'):\n                peek_hdf5(p)\n            else:\n                # まずは情報だけ\n                size = p.stat().st_size\n                print(f'\\n--- File info: {p} ---')\n                print('suffix:', suf, 'size:', human_size(size))\n        except Exception as e:\n            print(f'!! failed to peek {p}: {e}')\n\n# 1) まず全体を列挙して、何が入ってるか把握\nlist_tree(DATA_DIR, max_items=200)\n\n# 2) よく使う拡張子を中心に候補ファイルを拾う\ncand_ext = {'.csv', '.parquet', '.jsonl', '.ndjson', '.npz', '.npy', '.h5', '.hdf5'}\ncands = [p for p in DATA_DIR.rglob('*') if p.is_file() and p.suffix.lower() in cand_ext]\ncands.sort(key=lambda p: p.stat().st_size, reverse=True)\n\nprint('\\n=== candidate structured files (top) ===')\nfor i, p in enumerate(cands[:50], 1):\n    print(f'{i:02d}  {human_size(p.stat().st_size)}  {p}')\n\n# 3) “submission/train/test/metadata”っぽいものを優先的に覗く\nchoose_and_peek(cands)\n\n# 4) 最後に、推定を出すためのチェックリスト（手動で埋める）\nprint('\\n=== Checklist (fill after peeking) ===')\nprint('- submission columns: ???')\nprint('- submission rows (should match test units): ???')\nprint('- prediction unit: record? window? segment? ???')\nprint('- label type: binary / multi-class / multi-label ???')\nprint('- raw signal container: (npz/parquet/hdf5/...) ???')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-02T19:34:32.555675Z","iopub.execute_input":"2026-02-02T19:34:32.555991Z","iopub.status.idle":"2026-02-02T19:34:34.323438Z","shell.execute_reply.started":"2026-02-02T19:34:32.555966Z","shell.execute_reply":"2026-02-02T19:34:34.322619Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nimport pandas as pd\n\nBASE = Path('/kaggle/input/hms-harmful-brain-activity-classification')\n\n# 1) test.csv の全体行数（＝提出すべき eeg_id の数の目安）\ntest_path = BASE / 'test.csv'\ntest_df = pd.read_csv(test_path)\n\nprint('=== test.csv ===')\nprint('shape:', test_df.shape)\nprint('columns:', list(test_df.columns))\nprint(test_df.head(5))\nprint('unique eeg_id:', test_df['eeg_id'].nunique() if 'eeg_id' in test_df.columns else 'N/A')\nprint('unique spectrogram_id:', test_df['spectrogram_id'].nunique() if 'spectrogram_id' in test_df.columns else 'N/A')\n\n\n# 2) train_eegs の parquet を1本だけ開いて構造確認\neeg_dir = BASE / 'train_eegs'\nparquets = sorted(eeg_dir.glob('*.parquet'))\n\nprint('\\n=== train_eegs parquet files ===')\nprint('count:', len(parquets))\nprint('first 5:', [p.name for p in parquets[:5]])\n\nif not parquets:\n    raise RuntimeError('No parquet files found in train_eegs')\n\nsample_pq = parquets[0]\neeg_id = sample_pq.stem\n\nprint('\\n=== sample parquet ===')\nprint('file:', sample_pq)\nprint('eeg_id (from filename):', eeg_id)\n\ndf = pd.read_parquet(sample_pq)\n\nprint('shape:', df.shape)\nprint('columns:', list(df.columns))\nprint(df.head(3))\n\n# ざっくり欠損と型を見る（重くならない範囲）\nprint('\\n=== dtypes (top 20) ===')\nprint(df.dtypes.head(20))\n\nprint('\\n=== missing ratio (top 20) ===')\nmiss = (df.isna().mean()).sort_values(ascending=False)\nprint(miss.head(20))\n\n# 時系列の手がかりがあるか（indexや時間列）\nprint('\\n=== index info ===')\nprint('index type:', type(df.index))\ntry:\n    print('index name:', df.index.name)\nexcept Exception:\n    pass","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-02T19:45:39.383204Z","iopub.execute_input":"2026-02-02T19:45:39.384379Z","iopub.status.idle":"2026-02-02T19:45:40.040247Z","shell.execute_reply.started":"2026-02-02T19:45:39.384346Z","shell.execute_reply":"2026-02-02T19:45:40.039055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom pathlib import Path\n\nBASE = Path('/kaggle/input/hms-harmful-brain-activity-classification')\ntrain = pd.read_csv(BASE / 'train.csv')\n\ng = train.groupby('eeg_id').agg(\n    n_rows=('eeg_sub_id', 'size'),\n    n_sub=('eeg_sub_id', 'nunique'),\n    off_min=('eeg_label_offset_seconds', 'min'),\n    off_max=('eeg_label_offset_seconds', 'max'),\n    n_cons=('expert_consensus', 'nunique'),\n).reset_index()\n\n# 観察に向いた候補。\n# 1) subが多い。\n# 2) offsetが広い。\n# 3) consensusが混ざっている。\ncand = g.sort_values(\n    by=['n_sub', 'off_max', 'n_cons', 'n_rows'],\n    ascending=[False, False, False, False]\n)\n\nprint('=== top candidates (good for anatomy) ===')\nprint(cand.head(20))\n\n# いちおう、極端に小さいものを避けるフィルタ版も出す。\ncand2 = cand[(cand['n_sub'] >= 5) & ((cand['off_max'] - cand['off_min']) >= 10)]\nprint('\\n=== filtered candidates (n_sub>=5 and offset_span>=10s) ===')\nprint(cand2.head(20))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-02T19:51:06.870634Z","iopub.execute_input":"2026-02-02T19:51:06.871097Z","iopub.status.idle":"2026-02-02T19:51:07.087503Z","shell.execute_reply.started":"2026-02-02T19:51:06.871067Z","shell.execute_reply":"2026-02-02T19:51:07.086904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom pathlib import Path\n\nBASE = Path('/kaggle/input/hms-harmful-brain-activity-classification')\ntrain = pd.read_csv(BASE / 'train.csv')\n\nEEG_ID = 1460778765\n\nsub = train[train['eeg_id'] == EEG_ID].copy()\n\nprint('=== basic info ===')\nprint('eeg_id:', EEG_ID)\nprint('rows:', len(sub))\nprint('unique sub_id:', sub['eeg_sub_id'].nunique())\nprint('offset min/max:', sub['eeg_label_offset_seconds'].min(),\n      sub['eeg_label_offset_seconds'].max())\n\nprint('\\n=== head (sub_id / offset / consensus) ===')\nprint(sub[['eeg_sub_id', 'eeg_label_offset_seconds', 'expert_consensus']].head(10))\n\nprint('\\n=== tail ===')\nprint(sub[['eeg_sub_id', 'eeg_label_offset_seconds', 'expert_consensus']].tail(10))\n\nprint('\\n=== vote summary (mean over windows) ===')\nvotes = ['seizure_vote','lpd_vote','gpd_vote','lrda_vote','grda_vote','other_vote']\nprint(sub[votes].mean())\n\nprint('\\n=== vote example (first 5 windows) ===')\nprint(sub[votes].head(5))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-02T19:54:15.702981Z","iopub.execute_input":"2026-02-02T19:54:15.703563Z","iopub.status.idle":"2026-02-02T19:54:15.848772Z","shell.execute_reply.started":"2026-02-02T19:54:15.703534Z","shell.execute_reply":"2026-02-02T19:54:15.847923Z"}},"outputs":[],"execution_count":null}]}