{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7423091,"sourceType":"datasetVersion","datasetId":4319032}],"dockerImageVersionId":30635,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\ndf_traincsv = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\ndf_traincsv.head()\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-23T19:44:02.938877Z","iopub.execute_input":"2024-01-23T19:44:02.939320Z","iopub.status.idle":"2024-01-23T19:44:03.677062Z","shell.execute_reply.started":"2024-01-23T19:44:02.939286Z","shell.execute_reply":"2024-01-23T19:44:03.675896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Instead of reading each parquet file, I'm loading a single pandas dataframe that contains all spectrograms parquet files concatenated, with an extra column with the spectrogram id.","metadata":{}},{"cell_type":"code","source":"df = pd.read_parquet('/kaggle/input/hms-spectrograms-in-a-single-dataframe/00_single_spectrograms_originals.parquet')\ncolumns = df.columns\nprint(df.shape)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-23T19:44:08.378891Z","iopub.execute_input":"2024-01-23T19:44:08.379294Z","iopub.status.idle":"2024-01-23T19:44:46.342749Z","shell.execute_reply.started":"2024-01-23T19:44:08.379259Z","shell.execute_reply":"2024-01-23T19:44:46.341601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#\n# Spectrograms with NaN's\n#\n\nidxs = (pd.isna(df)).any(axis=1)\nspecs_with_nan = df[idxs]['spectrogram_id'].unique()\nspecs_with_nan.shape","metadata":{"execution":{"iopub.status.busy":"2024-01-23T19:45:12.850943Z","iopub.execute_input":"2024-01-23T19:45:12.851341Z","iopub.status.idle":"2024-01-23T19:45:15.994596Z","shell.execute_reply.started":"2024-01-23T19:45:12.851310Z","shell.execute_reply":"2024-01-23T19:45:15.993508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#\n# Spectrograms where all values are NaN.\n#\n\nspecs_with_all_nans = np.array([], dtype=int)\nfor spec_id in specs_with_nan:\n    sub_spec = df.loc[df.spectrogram_id == spec_id][columns[2:]]\n    if pd.isna(sub_spec).all(axis=None):\n        specs_with_all_nans = np.append(specs_with_all_nans, spec_id)\n\nspecs_with_all_nans\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\"> <b>Result:</b> There are no spectrograms in train.csv where everything is NaN.</div>","metadata":{}},{"cell_type":"code","source":"#\n# Indices in df_traincsv where sub spectrogram has some NaN values.\n#\n\nidxs_sub_specs_with_nans = np.array([], dtype=int)\nfor spec_id in specs_with_nan:\n    items = df_traincsv.loc[df_traincsv.spectrogram_id == spec_id]\n    for idx in items.index:\n        item = df_traincsv.iloc[idx]\n        sub_spec = df.loc[(df.spectrogram_id == item.spectrogram_id)&\n                          (df.time >= item.spectrogram_label_offset_seconds)&\n                          (df.time < (item.spectrogram_label_offset_seconds + 600))][columns[2:]]\n        if pd.isna(sub_spec).any(axis=None):\n            idxs_sub_specs_with_nans = np.append(idxs_sub_specs_with_nans, idx)\n\nprint(f'Number of sub eegs with NaN in sub spectrogram: {len(idxs_sub_specs_with_nans)}')\nprint(f'Total number of rows in train set: {len(df_traincsv)}')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\"> <b>Result:</b> There are sub eeg's with NaN's in the corresponding sub spectrogram.</div>\n","metadata":{}}]}