{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom sklearn import *\nfrom statistics import quantiles\nfrom scipy import signal\n\npd.set_option('mode.chained_assignment', None)\n\np = '../input/hms-harmful-brain-activity-classification/'\ntrain = pd.read_csv(p+'train.csv')\ncol = ['eeg_id', 'spectrogram_id', 'seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote', 'expert_consensus']\ntrain = train[col]\ntrain = train.drop_duplicates(subset=['eeg_id']).reset_index(drop=True)\ntest = pd.read_csv(p+'test.csv')\nsub = pd.read_csv(p+'sample_submission.csv')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-18T12:48:36.152134Z","iopub.execute_input":"2024-01-18T12:48:36.152878Z","iopub.status.idle":"2024-01-18T12:48:36.346079Z","shell.execute_reply.started":"2024-01-18T12:48:36.152828Z","shell.execute_reply":"2024-01-18T12:48:36.344907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_parquet('../input/hms-harmful-brain-activity-classification/train_eegs/'+ str(train.eeg_id[0]) +'.parquet')[:10000][:10_000][4_000:6_000] #200 * 5\n_ = df.plot(subplots=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-18T12:48:36.348287Z","iopub.execute_input":"2024-01-18T12:48:36.349063Z","iopub.status.idle":"2024-01-18T12:48:42.060902Z","shell.execute_reply.started":"2024-01-18T12:48:36.349027Z","shell.execute_reply":"2024-01-18T12:48:42.059771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fs = 200  # Sampling frequency (Hz)\n\n# Define frequency bands of interest\ndelta_band = (0.5, 4, \"delta\")  # Delta frequency band (Hz)\ntheta_band = (4, 8, \"theta\")    # Theta frequency band (Hz)\nalpha_band = (8, 13, \"alpha\")   # Alpha frequency band (Hz)\nbeta_band = (13, 30, \"beta\")   # Beta frequency band (Hz)\ngamma_band = (30, 90, \"gamma\")\n\nall_bands = [delta_band, theta_band, alpha_band, beta_band, gamma_band]\n\n# Design bandpass filters for each frequency band\ndef bandpass_filter(data, fs, band):\n    nyquist = 0.5 * fs\n    low = band[0] / nyquist\n    high = band[1] / nyquist\n    b, a = signal.butter(2, [low, high], btype='band')\n    filtered_data = signal.filtfilt(b, a, data)\n    return np.sum(filtered_data**2) / len(filtered_data)\n\ndef calculate_rhythms(df):\n    output_df = dict()\n    \n    for col in df.columns:\n        rhythm_energies = dict()\n        for rhythm in all_bands:\n            rhythm_energies[rhythm[2]] = bandpass_filter(df[col], fs, rhythm)\n        output_df[col] = rhythm_energies\n    output_df = pd.DataFrame(output_df)\n    return output_df","metadata":{"execution":{"iopub.status.busy":"2024-01-18T12:48:42.062607Z","iopub.execute_input":"2024-01-18T12:48:42.063058Z","iopub.status.idle":"2024-01-18T12:48:42.075923Z","shell.execute_reply.started":"2024-01-18T12:48:42.063020Z","shell.execute_reply":"2024-01-18T12:48:42.074669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's calculate statistics over different records\nMeans are really differ","metadata":{}},{"cell_type":"code","source":"from collections.abc import Iterable   # import directly from collections for Python < 3.3\n\ndef get_stats_(_path_):\n    df = pd.read_parquet(_path_)\n    return df.describe()\n\ndef getStats(path, ids, idname):\n    # calculate basic statistics over all channels\n    if isinstance(ids, Iterable):\n        all_data = []\n\n        for id_ in tqdm(ids):\n            df = get_stats_(path + str(id_) + '.parquet')\n            all_data.append(df)\n        return pd.concat(all_data)\n    else:\n        return getStats(path, [ids], idname)\n\ndef normilize_df(df):\n    return (df - df.mean()) / df.std()\n\ndef getRhythms(path, ids, idname):\n    all_data = []\n    head = [idname] + [band_[2] for band_ in all_bands]\n    if isinstance(ids, Iterable):\n        for id_ in tqdm(ids):\n            try:\n                df = pd.read_parquet(path + str(id_) + '.parquet')\n                df = normilize_df(df)\n                head = [idname] + [band_[2] + \" \" + col for band_ in all_bands for col in df.columns]\n                df_data = [id_] + list(calculate_rhythms(df).values.reshape(-1))\n            except:\n                df_data = [id_] + [np.nan for i in range(len(all_bands)*len(df.columns))]\n            all_data.append(df_data[:])\n        return pd.DataFrame(all_data, columns=head)\n    else:\n        return getRhythms(path, [ids], idname)\n\ndf = getStats('../input/hms-harmful-brain-activity-classification/train_eegs/', train.eeg_id.values[:10], 'eeg_id')\n\ndf.loc[\"mean\"]","metadata":{"execution":{"iopub.status.busy":"2024-01-18T12:48:42.078621Z","iopub.execute_input":"2024-01-18T12:48:42.079037Z","iopub.status.idle":"2024-01-18T12:48:42.749819Z","shell.execute_reply.started":"2024-01-18T12:48:42.079005Z","shell.execute_reply":"2024-01-18T12:48:42.748967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"getRhythms('../input/hms-harmful-brain-activity-classification/train_eegs/', train.eeg_id.values[0], 'eeg_id')","metadata":{"execution":{"iopub.status.busy":"2024-01-18T12:48:42.751079Z","iopub.execute_input":"2024-01-18T12:48:42.751417Z","iopub.status.idle":"2024-01-18T12:48:42.913746Z","shell.execute_reply.started":"2024-01-18T12:48:42.751386Z","shell.execute_reply":"2024-01-18T12:48:42.912936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrainx = getRhythms('../input/hms-harmful-brain-activity-classification/train_eegs/', train.eeg_id.values, 'eeg_id')\ntestx = getRhythms('../input/hms-harmful-brain-activity-classification/test_eegs/', test.eeg_id.values, 'eeg_id')","metadata":{"execution":{"iopub.status.busy":"2024-01-18T12:48:42.914738Z","iopub.execute_input":"2024-01-18T12:48:42.915071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#trainx = pd.concat((trainx, trainx2), axis=1)\ntrainx.to_csv('trainx.csv', index=False) #Use output with different models no rerun required\nprint(trainx.shape)\n\n#testx = pd.concat((testx, testx2), axis=1)\ntrainx.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epsilon=10e-15\ndef kl_divergence(solution, submission, micro=True):\n    for col in solution.columns:\n        submission[col] = np.clip(submission[col], epsilon, 1 - epsilon)\n        y_nonzero_indices = solution[col] != 0\n        solution[col] = solution[col].astype(float)\n        solution.loc[y_nonzero_indices, col] = solution.loc[y_nonzero_indices, col] * np.log(solution.loc[y_nonzero_indices, col] / submission.loc[y_nonzero_indices, col])\n        solution.loc[~y_nonzero_indices, col] = 0\n        if micro:\n            return np.average(solution.sum(axis=1))\n        else:\n            return np.average(solution.mean())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xcol = [c for c in trainx.columns if c not in ['eeg_id','spectrogram_id', 'expert_consensus']]\nycol = [c for c in train.columns if c not in ['eeg_id','spectrogram_id', 'expert_consensus']]\n\ncd = {'Seizure':'seizure_vote', 'GPD':'gpd_vote', 'LRDA':'lrda_vote', 'Other':'other_vote', 'GRDA':'grda_vote', 'LPD':'lpd_vote'}\ntrain['expert_consensus'] = train['expert_consensus'].map(cd)\nfor i in range(len(train)):\n    c = train['expert_consensus'][i]\n    train[c][i] = train[c][i]+10 #adding weight to expert consensus\n\nysum = train[ycol].sum(axis=1) \nfor c in ycol:\n    train[c] = (train[c] / ysum).astype(np.float64)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x1, x2, y1, y2 = model_selection.train_test_split(trainx[xcol].fillna(0), train[ycol], test_size=0.3, random_state=11, stratify=train.expert_consensus)\nmodel = ensemble.ExtraTreesRegressor(n_estimators=400, max_depth=None, n_jobs=-1, random_state=1, verbose=0, max_features='sqrt')\nmodel.fit(x1, y1)\npred = pd.DataFrame(model.predict(x2), columns=ycol)\nysum = pred.sum(axis=1)\nfor c in pred.columns: pred[c] = (pred[c] / ysum).astype(np.float64)\nscore=kl_divergence(y2.reset_index(drop=True), pred)\nprint(score)\nmodel.fit(trainx[xcol].fillna(0), train[ycol])\nsub = pd.DataFrame(model.predict(testx[xcol].fillna(0)), columns=ycol)\nysum = sub.sum(axis=1)\nfor c in sub.columns: sub[c] = (sub[c] / ysum).astype(np.float64)\nsub['eeg_id'] = test['eeg_id']\nsub.to_csv('submission.csv', index=False)\nsub.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fe = pd.DataFrame({'features':xcol, 'importance': model.feature_importances_})\nfe = fe.sort_values(by=['importance'], ascending=False)[:40]\nfe.plot(kind='barh', x='features', y='importance')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ｈ𝐀𝑷𝑷𝓎 🇰𝗮𝘨𝘨🇱𝖎Ｎɢ 💯","metadata":{}}]}