{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n#import os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-28T15:12:37.053172Z","iopub.execute_input":"2024-01-28T15:12:37.053571Z","iopub.status.idle":"2024-01-28T15:12:37.45845Z","shell.execute_reply.started":"2024-01-28T15:12:37.053539Z","shell.execute_reply":"2024-01-28T15:12:37.457157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:12:37.460137Z","iopub.execute_input":"2024-01-28T15:12:37.460605Z","iopub.status.idle":"2024-01-28T15:12:38.053164Z","shell.execute_reply.started":"2024-01-28T15:12:37.460573Z","shell.execute_reply":"2024-01-28T15:12:38.052165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")\ndf_test = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:12:38.054972Z","iopub.execute_input":"2024-01-28T15:12:38.055415Z","iopub.status.idle":"2024-01-28T15:12:38.326789Z","shell.execute_reply.started":"2024-01-28T15:12:38.055384Z","shell.execute_reply":"2024-01-28T15:12:38.325573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploratory Analysis","metadata":{}},{"cell_type":"code","source":"df_train.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:12:38.328662Z","iopub.execute_input":"2024-01-28T15:12:38.329013Z","iopub.status.idle":"2024-01-28T15:12:38.357996Z","shell.execute_reply.started":"2024-01-28T15:12:38.328983Z","shell.execute_reply":"2024-01-28T15:12:38.356976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.describe()","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:12:38.360888Z","iopub.execute_input":"2024-01-28T15:12:38.362187Z","iopub.status.idle":"2024-01-28T15:12:38.528757Z","shell.execute_reply.started":"2024-01-28T15:12:38.362143Z","shell.execute_reply":"2024-01-28T15:12:38.527674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Lenght train set (composed by multiple rows for the same patient):\")\nprint(len(df_train))","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:12:38.530327Z","iopub.execute_input":"2024-01-28T15:12:38.531335Z","iopub.status.idle":"2024-01-28T15:12:38.536322Z","shell.execute_reply.started":"2024-01-28T15:12:38.531298Z","shell.execute_reply":"2024-01-28T15:12:38.535392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:12:38.630548Z","iopub.execute_input":"2024-01-28T15:12:38.631571Z","iopub.status.idle":"2024-01-28T15:12:38.646175Z","shell.execute_reply.started":"2024-01-28T15:12:38.631531Z","shell.execute_reply":"2024-01-28T15:12:38.645056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Analysis on data distribution (eegs, spectrograms,..)","metadata":{}},{"cell_type":"code","source":"print(\"Number of unique patients: \")\nprint((df_train[\"patient_id\"].nunique()))","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:12:39.054653Z","iopub.execute_input":"2024-01-28T15:12:39.055036Z","iopub.status.idle":"2024-01-28T15:12:39.065368Z","shell.execute_reply.started":"2024-01-28T15:12:39.055Z","shell.execute_reply":"2024-01-28T15:12:39.064342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of uniques eeg_id: \")\nprint((df_train[\"eeg_id\"].nunique()))","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:12:39.275127Z","iopub.execute_input":"2024-01-28T15:12:39.275559Z","iopub.status.idle":"2024-01-28T15:12:39.285827Z","shell.execute_reply.started":"2024-01-28T15:12:39.275523Z","shell.execute_reply":"2024-01-28T15:12:39.284329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of uniques spectogram_id: \")\nprint((df_train[\"spectrogram_id\"].nunique()))","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:12:39.493791Z","iopub.execute_input":"2024-01-28T15:12:39.494149Z","iopub.status.idle":"2024-01-28T15:12:39.501446Z","shell.execute_reply.started":"2024-01-28T15:12:39.494119Z","shell.execute_reply":"2024-01-28T15:12:39.500093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_patient = 20\ndf_patient_sample = df_train[df_train[\"patient_id\"]==df_train[\"patient_id\"].unique()[random_patient]]\nprint(\"Number of eeg and spectrogram for a sample patient: \")\nlen(df_patient_sample)","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:12:39.734771Z","iopub.execute_input":"2024-01-28T15:12:39.735126Z","iopub.status.idle":"2024-01-28T15:12:39.748669Z","shell.execute_reply.started":"2024-01-28T15:12:39.735096Z","shell.execute_reply":"2024-01-28T15:12:39.74729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of uniques eeg_id for a single patient:\")\nprint((df_patient_sample[\"eeg_id\"].nunique()))","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:12:40.027971Z","iopub.execute_input":"2024-01-28T15:12:40.028429Z","iopub.status.idle":"2024-01-28T15:12:40.034502Z","shell.execute_reply.started":"2024-01-28T15:12:40.028392Z","shell.execute_reply":"2024-01-28T15:12:40.033378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of uniques eeg_id for a single patient: \")\nprint((df_patient_sample[\"spectrogram_id\"].nunique()))","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:12:40.263078Z","iopub.execute_input":"2024-01-28T15:12:40.263517Z","iopub.status.idle":"2024-01-28T15:12:40.270355Z","shell.execute_reply.started":"2024-01-28T15:12:40.263481Z","shell.execute_reply":"2024-01-28T15:12:40.268957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"for each patient there should be same number of unique eeg id and spectrogram id since the doctor takes those inputs simultaneosly to perform the prediction. The data says that for some patient the number of unique elemts is not the same (repetition of spectrograms for unique eegs?) No, probably the id of the spectogram remain the same even if it is shifted while the eeg change ids.","metadata":{}},{"cell_type":"code","source":"def no_match_eeg_spect_number(df):\n    count=0\n    df_no_match = pd.DataFrame()\n    for i in df[\"patient_id\"]:\n        df_patient_sample = df_train[df_train[\"patient_id\"]==i]\n        if((df_patient_sample[\"eeg_id\"].nunique())!=(df_patient_sample[\"spectrogram_id\"].nunique())):\n            count  += 1\n            df_no_match = pd.concat([df_no_match, df_patient_sample], ignore_index = True)\n            if(count==10):\n                return df_no_match\n    return df_no_match","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:12:40.796295Z","iopub.execute_input":"2024-01-28T15:12:40.796682Z","iopub.status.idle":"2024-01-28T15:12:40.803573Z","shell.execute_reply.started":"2024-01-28T15:12:40.796647Z","shell.execute_reply":"2024-01-28T15:12:40.802384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_no_match_number = no_match_eeg_spect_number(df_train)\nlen(df_no_match_number)","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:12:41.104929Z","iopub.execute_input":"2024-01-28T15:12:41.105299Z","iopub.status.idle":"2024-01-28T15:12:41.154449Z","shell.execute_reply.started":"2024-01-28T15:12:41.105261Z","shell.execute_reply":"2024-01-28T15:12:41.153663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's group the dataframe by the spectogram_id: if the number of unique eeg_ids associated to a single spectrogram id is greater than one, it means the spectrogram is duplicated but let's check if it is shifted. If the unique shifting values of the spectrogram is 3 (0,4,8 for example) but the eeg_id associated to the spectrogram are 4, than one of these samples is duplicated.","metadata":{}},{"cell_type":"code","source":"df_grouped_spectogram_id = df_no_match_number.groupby(\"spectrogram_id\")[[\"eeg_id\",\"spectrogram_label_offset_seconds\"]].nunique()\ndf_grouped_spectogram_id_more_than_two = df_grouped_spectogram_id[(df_grouped_spectogram_id[\"eeg_id\"] > 1) & (df_grouped_spectogram_id[\"spectrogram_label_offset_seconds\"] < df_grouped_spectogram_id[\"eeg_id\"] ) ]","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:12:41.736673Z","iopub.execute_input":"2024-01-28T15:12:41.737682Z","iopub.status.idle":"2024-01-28T15:12:41.751885Z","shell.execute_reply.started":"2024-01-28T15:12:41.737642Z","shell.execute_reply":"2024-01-28T15:12:41.750574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_grouped_spectogram_id_more_than_two","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:12:42.132935Z","iopub.execute_input":"2024-01-28T15:12:42.133317Z","iopub.status.idle":"2024-01-28T15:12:42.142424Z","shell.execute_reply.started":"2024-01-28T15:12:42.133273Z","shell.execute_reply":"2024-01-28T15:12:42.141306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"since the dataframe is empty, it means that the duplicated spectrogram id for a single eeg are not the same but they are shifted","metadata":{}},{"cell_type":"code","source":"print(\"Mean number of eegs per patient: \")\nprint(df_train.groupby(\"patient_id\")[\"eeg_id\"].nunique().mean())\n\nprint(\"Mean number of spectrogram per patient: \")\nprint(df_train.groupby(\"patient_id\")[\"spectrogram_id\"].nunique().mean())","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:12:42.759184Z","iopub.execute_input":"2024-01-28T15:12:42.759873Z","iopub.status.idle":"2024-01-28T15:12:42.784344Z","shell.execute_reply.started":"2024-01-28T15:12:42.759837Z","shell.execute_reply":"2024-01-28T15:12:42.783408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(df_train.groupby(\"patient_id\")[\"eeg_id\"].nunique())\nplt.xlim([-1,100])","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:12:44.78725Z","iopub.execute_input":"2024-01-28T15:12:44.787631Z","iopub.status.idle":"2024-01-28T15:12:46.050552Z","shell.execute_reply.started":"2024-01-28T15:12:44.787596Z","shell.execute_reply":"2024-01-28T15:12:46.047694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(df_train.groupby(\"patient_id\")[\"spectrogram_id\"].nunique())\nplt.xlim([-1,100])","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:12:46.052395Z","iopub.execute_input":"2024-01-28T15:12:46.053389Z","iopub.status.idle":"2024-01-28T15:12:46.542641Z","shell.execute_reply.started":"2024-01-28T15:12:46.053344Z","shell.execute_reply":"2024-01-28T15:12:46.541898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Analysis on divergent predictions considering the sampe patient","metadata":{}},{"cell_type":"markdown","source":"For a single patient, the different eeg and spectrogram input data will lead to the same conclusion ?","metadata":{}},{"cell_type":"code","source":"random_patient=1\ndf_sample_patient = df_train[df_train[\"patient_id\"]==df_train[\"patient_id\"].unique()[random_patient]]","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:15:43.196889Z","iopub.execute_input":"2024-01-28T15:15:43.19728Z","iopub.status.idle":"2024-01-28T15:15:43.20522Z","shell.execute_reply.started":"2024-01-28T15:15:43.197225Z","shell.execute_reply":"2024-01-28T15:15:43.204119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = df_sample_patient.columns.values[-6:len(df_sample_patient.columns)]\nprint(targets)\ntargets_with_eeg=np.append(targets,\"eeg_id\")","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:15:43.423915Z","iopub.execute_input":"2024-01-28T15:15:43.424422Z","iopub.status.idle":"2024-01-28T15:15:43.431995Z","shell.execute_reply.started":"2024-01-28T15:15:43.424381Z","shell.execute_reply":"2024-01-28T15:15:43.430943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_1_if_remain_fixed = df_sample_patient.groupby(\"eeg_id\")[targets].mean()/len(df_sample_patient.groupby(\"eeg_id\"))\n#df_1_if_remain_fixed[(df_1_if_remain_fixed[targets]!=0) & (df_1_if_remain_fixed[targets]!=1)]\ndf_1_if_remain_fixed","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:15:43.813807Z","iopub.execute_input":"2024-01-28T15:15:43.81419Z","iopub.status.idle":"2024-01-28T15:15:43.83832Z","shell.execute_reply.started":"2024-01-28T15:15:43.814156Z","shell.execute_reply":"2024-01-28T15:15:43.837259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_sample_patient[\"eeg_id\"].nunique())\nprint(df_sample_patient[\"spectrogram_id\"].nunique())\nsns.histplot(df_sample_patient[\"expert_consensus\"])","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:12:56.441567Z","iopub.execute_input":"2024-01-28T15:12:56.44196Z","iopub.status.idle":"2024-01-28T15:12:56.649602Z","shell.execute_reply.started":"2024-01-28T15:12:56.441925Z","shell.execute_reply":"2024-01-28T15:12:56.648516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def detect_contrast_opinion_same_patient(df):\n    contrast_opinion=0\n    for i in df[\"patient_id\"]:\n        df_patient_sample = df[df[\"patient_id\"]==i]\n        if(df_patient_sample[\"expert_consensus\"].nunique()!=1):\n            contrast_opinion +=1\n    return contrast_opinion/len(df_train)\n    ","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:12:56.891062Z","iopub.execute_input":"2024-01-28T15:12:56.89148Z","iopub.status.idle":"2024-01-28T15:12:56.897573Z","shell.execute_reply.started":"2024-01-28T15:12:56.891443Z","shell.execute_reply":"2024-01-28T15:12:56.896336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"percent_contrast_ides = detect_contrast_opinion_same_patient(df_train)\nprint(percent_contrast_ides)","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:12:57.965097Z","iopub.execute_input":"2024-01-28T15:12:57.965499Z","iopub.status.idle":"2024-01-28T15:14:06.969996Z","shell.execute_reply.started":"2024-01-28T15:12:57.965464Z","shell.execute_reply":"2024-01-28T15:14:06.968816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = df_train.columns.values[-6:len(df_train.columns)]\nprint(targets)","metadata":{"execution":{"iopub.status.busy":"2024-01-28T15:14:06.971759Z","iopub.execute_input":"2024-01-28T15:14:06.972083Z","iopub.status.idle":"2024-01-28T15:14:06.977759Z","shell.execute_reply.started":"2024-01-28T15:14:06.972054Z","shell.execute_reply":"2024-01-28T15:14:06.976612Z"},"trusted":true},"execution_count":null,"outputs":[]}]}