{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":8065485,"sourceType":"datasetVersion","datasetId":4758335},{"sourceId":8065528,"sourceType":"datasetVersion","datasetId":4758363},{"sourceId":8065751,"sourceType":"datasetVersion","datasetId":4758536},{"sourceId":8065920,"sourceType":"datasetVersion","datasetId":4758657}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import subprocess\nsubprocess.run('conda install -c conda-forge r-base', shell=True)\n!pip install dcor\n!pip install rpy2\n\n\n#subprocess.run('conda install --offline /input/rbasetar/R-4.3.3', shell=True)\n#! pip install ../input/rbasetar/R-4.3.3\n#! pip install ../input/dcorwhl/dcor-0.6-py3-none-any.whl\n#! pip install ../input/rpytar/rpy2-3.5.16","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#DWT\n\n#Wavelet spectra method for extracting features from signal data. Adapted methods from Vimalajeewa, Bruce, and Vidakovic for extracting wavelet spectra\n#slopes of DWT energies from protein mass spectra using distance-variance calculation of wavelet coefficients for each dyadic level. In this\n#application, OLS estimates of slopes for EEG DWT energies was not robust enough to accurately estimate slopes. Thus, distance variance energies for\n#dyadic levels 5-9 are extracted and used as features with better success in terms of predictive accuracy.\n\n#Vimalajeewa D, Bruce SA, Vidakovic B. Early detection of ovarian cancer by wavelet analysis of protein mass spectra.\n#   Stat Med. 2023 Jun 15;42(13):2257-2273. doi: 10.1002/sim.9722. Epub 2023 Mar 31. PMID: 36999745.\n\n\nimport numpy as np\nimport pandas as pd\nimport copy\nimport dcor\nimport rpy2\nimport rpy2.robjects as ro\nfrom rpy2.robjects.packages import importr\nfrom rpy2.robjects import pandas2ri\n\nmeta = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nmetatest = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\n\ndef dwtra(data, h, L):  #Created on 5-19-2019 by Brani Vidakovic, modified 3-17-2021 by Emory Fields\n    n = len(h)\n    C = data\n    dwtra = np.array([])\n    H = h\n    G = copy.deepcopy(H)\n    G = np.resize(G, (1, len(G)))\n    G = np.fliplr(G)\n    G = np.resize(G[0], (len(G[0],)))\n    G[np.arange(0, n, 2)] = -G[np.arange(0, n, 2)] \n    for j in range(int(L)): \n        nn = len(C)\n        C = np.concatenate((C[np.mod(np.arange(-(n-1), 0), nn)], C, C))\n        D = np.convolve(C, G)\n        D = D[np.arange(n, n+nn - 1, 2)]\n        C = np.convolve(C, H)\n        C = C[np.arange(n - 1, n+nn - 2, 2) + 1]\n        dwtra = np.concatenate((D, dwtra))\n    dwtra = np.concatenate((C,dwtra))\n    return dwtra\n\ndef waveletspectra(data, wf):  #Created on 3-12-2021 by Brani Vidakovic, modified 3/11/2024 by harved3\n            \n    lnn = np.log2(len(data))\n    wddata = dwtra(data, h = np.array(wf), L = lnn-1)\n    y = np.array([])\n    for i in np.arange(1, int(lnn)):\n        help1 = wddata[int(round(2**i)) : int(round(2**(i + 1)))]\n        dv1 = dcor.distance_covariance_sqr(help1,help1)\n        dv1 = np.array([dv1])\n        y = np.concatenate((y, dv1))\n    log2spec = np.log2(y)\n\n    return(log2spec)\n\ndaub4 = [-0.010597401784997278, 0.032883011666982945, \\\n0.030841381835986965, -0.18703481171888114, \\\n-0.02798376941698385, 0.6308807679295904, \\\n0.7148465705525415, 0.23037781330885523]\n\n### DWT of training data ###\n\nframes = []\n\nfor i in range(100):   #discrete wavelet transform and extraction of log2 energies\n    energies = [] #empty list to store wavelet energies\n    offset = int(0) #initialize moving window\n\n    eeg_idx = meta.iloc[i,0]    #eeg id to search for parquet file\n    eeg_offset = meta.iloc[i,2] #eeg offset (seconds)\n    offsetx = int(eeg_offset*200+3976)   #row corresponding to start of the central 10 seconds of eeg (200 data points per second)\n    offsetx_end = int(offsetx+2048)    #row corresponding to end of the central 10 seconds of eeg (200 data points per second)\n    eegx = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/{}.parquet'.format(eeg_idx), engine='pyarrow')\n    eegx_sub = eegx[offsetx:offsetx_end]    #subset of the middle 2048 rows for the 50-second eeg\n\n    if eegx_sub.isnull().values.any():\n        eegx_sub = eegx_sub.ffill()     #forward fill missing values\n\n    vector = eegx_sub.values.flatten('F')   #flatten dataframe to a single row for easier processing\n\n    for j in range(20):    \n        xx = vector[offset:offset+2048] #window for DWT\n        en = waveletspectra(xx, daub4)\n        energies.append(en[range(4,9)])\n\n        offset += 2048 #move to the next EEG node\n\n    energies = np.concatenate(energies)\n    frames.append(energies)\n\ndvdaub4 = pd.DataFrame(frames)\ncon = meta[\"expert_consensus\"]\ndvdaub4 = pd.concat([dvdaub4,con],axis=1)\n\n### DWT of test data\n\nframest = []\n\nfor i in range(len(metatest)):   #discrete wavelet transform and extraction of log2 energies\n    energiest = [] #empty list to store wavelet energies\n    offsett = int(0) #initialize moving window\n\n    eeg_idxt = metatest.iloc[i,1]    #eeg id to search for parquet file\n    offsetxt = int(3976)   #row corresponding to start of the central 10 seconds of eeg (200 data points per second)\n    offsetx_endt = int(offsetxt+2048)    #row corresponding to end of the central 10 seconds of eeg (200 data points per second)\n    eegxt = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/{}.parquet'.format(eeg_idxt), engine='pyarrow')\n    eegx_subt = eegxt[offsetxt:offsetx_endt]    #subset of the middle 2048 rows for the 50-second eeg\n\n    if eegx_subt.isnull().values.any():\n        eegx_subt = eegx_subt.ffill()     #forward fill missing values\n\n    vectort = eegx_subt.values.flatten('F')   #flatten dataframe to a single row for easier processing\n\n    for j in range(20):    \n        xxt = vectort[offsett:offsett+2048] #window for DWT\n        ent = waveletspectra(xxt, daub4)\n        energiest.append(ent[range(4,9)])\n\n        offsett += 2048 #move to the next EEG node\n\n    energiest = np.concatenate(energiest)\n    framest.append(energiest)\n\ntest = pd.DataFrame(framest)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-04-08T15:16:13.632147Z","iopub.execute_input":"2024-04-08T15:16:13.632549Z","iopub.status.idle":"2024-04-08T15:17:07.548886Z","shell.execute_reply.started":"2024-04-08T15:16:13.632516Z","shell.execute_reply":"2024-04-08T15:17:07.547428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#R code\n\nutils = importr('utils')\nbase = importr('base')\nutils.chooseCRANmirror(ind=1)\n\nutils.install_packages('caret')\n\nwith (ro.default_converter + pandas2ri.converter).context(): \n    dvdaub4r = ro.conversion.get_conversion().py2rpy(dvdaub4) #convert python dataframe to R\n    \nwith (ro.default_converter + pandas2ri.converter).context():\n    testr = ro.conversion.get_conversion().py2rpy(test) #convert python dataframe to R\n\n#caret package in R for knn fitting (train/test split CV previously performed on training data with 87% test accuracy at k=5)\nknnpred = ro.r(\"\"\"\n        fit <- function(train, test){\n            require(caret)\n            eegnames = c(paste(rep(c('Fp1', 'F3', 'C3', 'P3', 'F7', 'T3', 'T5', 'O1', 'Fz', 'Cz',\n            'Pz', 'Fp2', 'F4', 'C4', 'P4', 'F8', 'T4', 'T6', 'O2', 'EKG'),each=5),\n            rep(c(1:5),20),sep=\".\"),\"con\")\n\n            colnames(train) = eegnames\n\n            train[sapply(train,is.infinite)] <- NA\n            train = train[complete.cases(train),]\n    \n            colnames(test) = eegnames[-101]\n\n            train$con = as.factor(train$con)\n            train[,-101] <- scale(train[,-101])\n\n            set.seed(123)\n            fit.knn <- train(con~., data = train, method = \"knn\", metric = \"Accuracy\",\n                trControl=trainControl(method=\"repeatedcv\", number=10, repeats=1))\n\n            set.seed(123)\n            prediction <- predict(fit.knn, newdata=test, type=\"prob\")\n            return(prediction)\n    }\"\"\")\n\n\nr_fit = ro.r['fit']\npreds = r_fit(dvdaub4r,testr) #python function to call the R function!\n\neeg_id = metatest[\"eeg_id\"]\nsubmission = pd.DataFrame(preds)\nsubmission = submission.T\nsubmission = pd.concat([submission,eeg_id],axis=1)\nsubmission = submission.rename(columns={0: \"gpd_vote\", 1: \"grda_vote\", 2: \"lpd_vote\", 3: \"lrda_vote\", 4: \"other_vote\", 5: \"seizure_vote\"})\nsubmission = submission[['eeg_id', 'seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']]\n\nsubmission.to_csv('submission.csv', index=False)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-04-08T09:56:18.485921Z","iopub.execute_input":"2024-04-08T09:56:18.487776Z","iopub.status.idle":"2024-04-08T13:37:57.001180Z","shell.execute_reply.started":"2024-04-08T09:56:18.487719Z","shell.execute_reply":"2024-04-08T13:37:56.996404Z"},"trusted":true},"execution_count":null,"outputs":[]}]}