{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport gc\nimport re\nimport sys\nimport time\nimport json\nimport numba\nimport joblib\nimport random\nfrom collections import defaultdict, Counter\nfrom itertools import combinations, permutations\nfrom functools import reduce, partial\nimport numpy as np\nimport pandas as pd\nimport lightgbm as lgb\nimport xgboost as xgb\nimport catboost as cbt\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport plotly\nimport plotly.express as px\nimport plotly.graph_objects as go\n\nimport librosa\n\nfrom sklearn.model_selection import KFold, GroupKFold, StratifiedKFold, StratifiedGroupKFold, TimeSeriesSplit\nfrom sklearn.metrics import roc_auc_score, accuracy_score, recall_score, average_precision_score, auc, f1_score, precision_recall_curve\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score, mean_squared_log_error, log_loss\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler, RobustScaler\nfrom sklearn.preprocessing import LabelEncoder\nlb = LabelEncoder()\n\ntry:\n    import torch\n    from torch import nn\n    from torch import optim\n    import torch.nn.functional as F\n    from torch.utils.data import DataLoader, Dataset\n\n    is_torch_available = True\nexcept ImportError:\n    is_torch_available = False\ntry:\n    import pytorch_lightning as pl\n\n    is_pl_available = True\nexcept ImportError:\n    is_pl_available = False\ntry:\n    import tensorflow as tf\n    \n    is_tf_available = True\nexcept ImportError:\n    is_tf_available = False\n\nimport tqdm\nfrom tqdm.auto import tqdm as auto_tqdm\nfrom tqdm.notebook import tqdm as nb_tqdm\nnb_tqdm.pandas()\n\nimport warnings\nwarnings.filterwarnings('ignore')\n# warnings.simplefilter(action=\"ignore\", category=pd.errors.PerformanceWarning)\n\nkaggle_keep_running = lambda t: [time.sleep(1) for _ in auto_tqdm(range(t))]","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-06T06:59:56.889978Z","iopub.execute_input":"2024-02-06T06:59:56.890608Z","iopub.status.idle":"2024-02-06T07:00:24.382518Z","shell.execute_reply.started":"2024-02-06T06:59:56.890574Z","shell.execute_reply":"2024-02-06T07:00:24.381417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.environ[\"CUDA_VISIBLE_DEVICES\"] = '0'\n\nclass CONFIG:\n    global tst_eegs_dir, trn_eegs_dir, tst_spec_dir, trn_spec_dir\n    \n    seed = 42\n    use_gpu = True\n    \n    root = '/kaggle/input/hms-harmful-brain-activity-classification'\n    tst_csv_path = root + '/test.csv'\n    trn_csv_path = root + '/train.csv'\n    sub_csv_path = root + '/sample_submission.csv'\n    tst_eegs_dir = root + '/test_eegs'\n    trn_eegs_dir = root + '/train_eegs'\n    tst_spec_dir = root + '/test_spectrograms'\n    trn_spec_dir = root + '/train_spectrograms'\n    tst_eegs_paths = list(map(lambda p: os.path.join(tst_eegs_dir, p), os.listdir(tst_eegs_dir)))\n    trn_eegs_paths = list(map(lambda p: os.path.join(trn_eegs_dir, p), os.listdir(trn_eegs_dir)))\n    tst_spec_paths = list(map(lambda p: os.path.join(tst_spec_dir, p), os.listdir(tst_spec_dir)))\n    trn_spec_paths = list(map(lambda p: os.path.join(trn_spec_dir, p), os.listdir(trn_spec_dir)))\n\n    is_torch_available = is_torch_available\n    is_pl_available = is_pl_available\n    is_tf_available = is_tf_available\n\n    is_debug = False\n    is_test = False\n    is_submission = False\n    is_gpu_support = os.system('nvidia-smi 1> /dev/null 2> /dev/null') == 0\n    if is_torch_available:\n        is_gpu_available = use_gpu and is_gpu_support and torch.cuda.is_available()\n        if is_gpu_available:\n            gpu_num = torch.cuda.device_count()\n            gpu_names = [torch.cuda.get_device_name(i) for i in range(gpu_num)]\n            gpu_cur_dev_idx = torch.cuda.current_device()\n    elif is_tf_available:\n        is_gpu_available = use_gpu and is_gpu_support and tf.test.is_gpu_available()\n        if is_gpu_available:\n            gpu_num = tf.config.list_physical_devices('GPU')\n            gpu_names = [tf.config.list_physical_devices('GPU')[i].name for i in range(gpu_num)]\n            gpu_cur_dev_idx = tf.config.list_physical_devices('GPU')[0].name\n    \n    ''' Pandas '''\n    pd_display_max_rows = 60  # default 60\n    pd_display_max_columns = 0  # default 0\n    pd_display_expand_frame_repr = True  # default True\n    pd_display_max_colwidth = 50  # default 50\n    pd_display_precision = 6  # default 6\n    pd.set_option('display.max_rows', pd_display_max_rows)\n    pd.set_option('display.max_columns', pd_display_max_columns)\n    pd.set_option('display.expand_frame_repr', pd_display_expand_frame_repr)\n    pd.set_option('display.max_colwidth', pd_display_max_colwidth)\n    pd.set_option('display.precision', pd_display_precision)","metadata":{"execution":{"iopub.status.busy":"2024-02-06T07:00:24.384363Z","iopub.execute_input":"2024-02-06T07:00:24.385070Z","iopub.status.idle":"2024-02-06T07:00:24.790018Z","shell.execute_reply.started":"2024-02-06T07:00:24.385042Z","shell.execute_reply":"2024-02-06T07:00:24.789151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tst_csv = pd.read_csv(CONFIG.tst_csv_path)\ntrn_csv = pd.read_csv(CONFIG.trn_csv_path)","metadata":{"execution":{"iopub.status.busy":"2024-02-06T07:58:43.026215Z","iopub.execute_input":"2024-02-06T07:58:43.026601Z","iopub.status.idle":"2024-02-06T07:58:43.179884Z","shell.execute_reply.started":"2024-02-06T07:58:43.026575Z","shell.execute_reply":"2024-02-06T07:58:43.178798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_non_overlapp_eeg_id_trn_csv(df):\n    TARGETS = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n    \n    train = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n        {'spectrogram_id': 'first', 'spectrogram_label_offset_seconds': 'min'}).astype(int)\n    train.columns = ['spec_id','min']\n\n    tmp = df.groupby('eeg_id')[['spectrogram_label_offset_seconds']].agg('max').astype(int)\n    train['max'] = tmp\n\n    tmp = df.groupby('eeg_id')[['patient_id']].agg('first').astype(int)\n    train['patient_id'] = tmp\n\n    tmp = df.groupby('eeg_id')[TARGETS].agg('sum')\n    for t in TARGETS:\n        train[t] = tmp[t].values\n\n    y_data = train[TARGETS].values\n    y_data = y_data / y_data.sum(axis=1, keepdims=True)\n    train[TARGETS] = y_data\n\n    tmp = df.groupby('eeg_id')[['expert_consensus']].agg('first')\n    train['target'] = tmp\n\n    train = train.reset_index()\n    print('Train non-overlapp eeg_id shape:', train.shape)\n    return train","metadata":{"execution":{"iopub.status.busy":"2024-02-06T07:58:43.260587Z","iopub.execute_input":"2024-02-06T07:58:43.260966Z","iopub.status.idle":"2024-02-06T07:58:43.269817Z","shell.execute_reply.started":"2024-02-06T07:58:43.260939Z","shell.execute_reply":"2024-02-06T07:58:43.268548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trn_csv = get_non_overlapp_eeg_id_trn_csv(trn_csv)\nif CONFIG.is_debug:\n    trn_csv = trn_csv.sample(10, ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2024-02-06T07:59:17.940502Z","iopub.execute_input":"2024-02-06T07:59:17.940887Z","iopub.status.idle":"2024-02-06T07:59:17.947019Z","shell.execute_reply.started":"2024-02-06T07:59:17.940858Z","shell.execute_reply":"2024-02-06T07:59:17.945863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trn_csv.to_csv('trn.csv', index=False)\ntrn_csv","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1. save origin spec","metadata":{}},{"cell_type":"code","source":"ori_spec = {}\nfor i in auto_tqdm(range(len(trn_csv))):\n    row = trn_csv.iloc[i]\n    r = int((row['min'] + row['max']) // 4)\n    p = f\"{trn_spec_dir}/{row['spec_id']}.parquet\"\n    ori_spec_np = pd.read_parquet(p).iloc[r:r+300, 1:].values  # df(300~3000, 401) -> np(300, 400)\n    ori_spec[row['spec_id']] = ori_spec_np","metadata":{"execution":{"iopub.status.busy":"2024-02-06T08:10:36.540385Z","iopub.execute_input":"2024-02-06T08:10:36.540757Z","iopub.status.idle":"2024-02-06T08:10:36.960011Z","shell.execute_reply.started":"2024-02-06T08:10:36.540729Z","shell.execute_reply":"2024-02-06T08:10:36.958666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.save('ori_spec.npy', ori_spec)","metadata":{"execution":{"iopub.status.busy":"2024-02-06T08:11:04.857608Z","iopub.execute_input":"2024-02-06T08:11:04.858569Z","iopub.status.idle":"2024-02-06T08:11:04.879072Z","shell.execute_reply.started":"2024-02-06T08:11:04.858532Z","shell.execute_reply":"2024-02-06T08:11:04.877714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. save eeg spec","metadata":{}},{"cell_type":"code","source":"def sign2spec(series):\n    # fill nan\n#     series = series.fillna(0)\n    series = series.fillna(series.mean(skipna=True))\n#     series = series.fillna(series.median(skipna=True))\n    \n    sign = series.values\n    sample_freq = 200\n    time_width = 300\n    freq_num = 100\n    spec = librosa.feature.melspectrogram(\n        y=sign, sr=sample_freq, hop_length=len(sign)//time_width,\n        n_fft=1024, n_mels=freq_num, fmin=0, fmax=20, win_length=128,\n    )\n        \n    return spec[:, :time_width].T  # (time_width=300, freq_num=100)\n\ndef eeg_df2spec_np(eeg_df):\n    eeg_df = eeg_df.copy()\n    bgn_idx = (len(eeg_df) - 10_000) // 2\n    end_idx = bgn_idx + 10_000\n    eeg_df = eeg_df.iloc[bgn_idx:end_idx]  # (10_000, 20)\n    spec_np = np.zeros((300, 400), dtype='float32')  # (300, 400)\n    spec_np[:, :100] = sign2spec(eeg_df['Fp1']-eeg_df['F7']) + sign2spec(eeg_df['F7']-eeg_df['T3']) + \\\n                       sign2spec(eeg_df['T3']-eeg_df['T5']) + sign2spec(eeg_df['T5']-eeg_df['O1'])  # LL\n    spec_np[:, 100:200] = sign2spec(eeg_df['Fp1']-eeg_df['F7']) + sign2spec(eeg_df['F7']-eeg_df['T3']) + \\\n                          sign2spec(eeg_df['T3']-eeg_df['T5']) + sign2spec(eeg_df['T5']-eeg_df['O1'])  # LP\n    spec_np[:, 200:300] = sign2spec(eeg_df['Fp1']-eeg_df['F7']) + sign2spec(eeg_df['F7']-eeg_df['T3']) + \\\n                          sign2spec(eeg_df['T3']-eeg_df['T5']) + sign2spec(eeg_df['T5']-eeg_df['O1'])  # RL\n    spec_np[:, 300:] = sign2spec(eeg_df['Fp1']-eeg_df['F7']) + sign2spec(eeg_df['F7']-eeg_df['T3']) + \\\n                       sign2spec(eeg_df['T3']-eeg_df['T5']) + sign2spec(eeg_df['T5']-eeg_df['O1'])  # RP\n    spec_np = spec_np / 4\n    return spec_np","metadata":{"execution":{"iopub.status.busy":"2024-02-06T08:26:01.968047Z","iopub.execute_input":"2024-02-06T08:26:01.968582Z","iopub.status.idle":"2024-02-06T08:26:01.981623Z","shell.execute_reply.started":"2024-02-06T08:26:01.968462Z","shell.execute_reply":"2024-02-06T08:26:01.980344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eeg_spec = {}\nfor i in auto_tqdm(range(len(trn_csv))):\n    row = trn_csv.iloc[i]\n    p = f\"{trn_eegs_dir}/{row['eeg_id']}.parquet\"\n    eeg_spec_np = eeg_df2spec_np(pd.read_parquet(p))  # df(10_000~684_400, 20) -> np(300, 400)\n    eeg_spec[row['eeg_id']] = eeg_spec_np","metadata":{"execution":{"iopub.status.busy":"2024-02-06T08:26:12.362518Z","iopub.execute_input":"2024-02-06T08:26:12.363469Z","iopub.status.idle":"2024-02-06T08:26:14.314299Z","shell.execute_reply.started":"2024-02-06T08:26:12.363431Z","shell.execute_reply":"2024-02-06T08:26:14.312790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.save('eeg_spec.npy', eeg_spec)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}