{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"! pip install scipy\n! pip install scikit-multilearn","metadata":{"execution":{"iopub.status.busy":"2024-04-24T06:07:12.016611Z","iopub.execute_input":"2024-04-24T06:07:12.016997Z","iopub.status.idle":"2024-04-24T06:08:18.976412Z","shell.execute_reply.started":"2024-04-24T06:07:12.016934Z","shell.execute_reply":"2024-04-24T06:08:18.974677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport pandas as pd \nimport numpy as np \nimport os\n\nfrom sklearn.metrics import mean_squared_error, mean_absolute_error, r2_score\nfrom sklearn.preprocessing import RobustScaler\nfrom scipy.signal import welch\nfrom zlib import crc32","metadata":{"execution":{"iopub.status.busy":"2024-04-24T06:08:18.979320Z","iopub.execute_input":"2024-04-24T06:08:18.979881Z","iopub.status.idle":"2024-04-24T06:08:20.449358Z","shell.execute_reply.started":"2024-04-24T06:08:18.979822Z","shell.execute_reply":"2024-04-24T06:08:20.448013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training dataset (main dataset)\ndf = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\ndf.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T06:08:20.451043Z","iopub.execute_input":"2024-04-24T06:08:20.451802Z","iopub.status.idle":"2024-04-24T06:08:20.732747Z","shell.execute_reply.started":"2024-04-24T06:08:20.451747Z","shell.execute_reply":"2024-04-24T06:08:20.731478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# using an id to check the main dataset \nfilter_df = df[df['eeg_id'] == 722738444]\nfilter_df","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:00:35.130602Z","iopub.execute_input":"2024-04-07T11:00:35.131026Z","iopub.status.idle":"2024-04-07T11:00:35.151608Z","shell.execute_reply.started":"2024-04-07T11:00:35.130997Z","shell.execute_reply":"2024-04-07T11:00:35.150240Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training dataset (sub dataset - eeg records for just one eeg_id)\neeg_sample = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/722738444.parquet')\neeg_sample","metadata":{"execution":{"iopub.status.busy":"2024-04-06T09:22:41.654473Z","iopub.execute_input":"2024-04-06T09:22:41.654835Z","iopub.status.idle":"2024-04-06T09:22:41.821586Z","shell.execute_reply.started":"2024-04-06T09:22:41.654804Z","shell.execute_reply":"2024-04-06T09:22:41.820657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training dataset (sub dataset - eeg records for just one eeg_id)\neeg_sample = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1001369401.parquet')\neeg_sample","metadata":{"execution":{"iopub.status.busy":"2024-04-07T11:01:09.648836Z","iopub.execute_input":"2024-04-07T11:01:09.649283Z","iopub.status.idle":"2024-04-07T11:01:09.694028Z","shell.execute_reply.started":"2024-04-07T11:01:09.649249Z","shell.execute_reply":"2024-04-07T11:01:09.692903Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training dataset (sub dataset - spectrograms records for just one spectrograms_id)\nspectrograms_sample = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/1003735183.parquet')\nspectrograms_sample","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sum the voting count across columns for each row, group by eeg_id\nvoting_count = df.groupby('eeg_id')[['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote','other_vote']].sum()\nvoting_count['vote_sum'] = voting_count.sum(axis=1)\nvoting_count[\"row_count\"] = df.groupby('eeg_id').size()\nvoting_count[\"averge_vote_sum\"] = round(voting_count['vote_sum'] / voting_count[\"row_count\"]).round(0).astype(int)\n\n# calculate the total sum of the 'sum' column and find the avg\ntotal_voting = voting_count['vote_sum'].sum()\nnum_ids = len(voting_count)\navg = round(total_voting / num_ids)\nprint(\"Average number of total ppl voting:\", avg)\n\n# calculate the avg 'sum' based on 'row_count'\navg_sum_by_row_count = voting_count.groupby('row_count')['vote_sum'].mean().round(0).reset_index().astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-04-09T08:57:01.740764Z","iopub.execute_input":"2024-04-09T08:57:01.741845Z","iopub.status.idle":"2024-04-09T08:57:01.808545Z","shell.execute_reply.started":"2024-04-09T08:57:01.741775Z","shell.execute_reply":"2024-04-09T08:57:01.807168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.0 Dataset Transformation\n\nGiven that the ML model aims to classify different EEG patterns such as Seizure, LPD, GPD, LRD, and GRDA based on electrode readings (Fp1, Cz, Pz, etc.), it becomes crucial to balance the need for dimensionality reduction with the preservation of meaningful information for accurate classification. Here are some actions taken: -","metadata":{}},{"cell_type":"code","source":"# cleaning the main training dataset by dropping data duplication and unwanted columns\ndf_v1 = df.drop(columns = ['eeg_sub_id', 'eeg_label_offset_seconds', 'spectrogram_sub_id', 'spectrogram_label_offset_seconds', 'label_id'])\ndf_v1 = df_v1.drop_duplicates()\ndf_v1.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-04-09T08:57:05.742033Z","iopub.execute_input":"2024-04-09T08:57:05.742662Z","iopub.status.idle":"2024-04-09T08:57:05.811958Z","shell.execute_reply.started":"2024-04-09T08:57:05.742614Z","shell.execute_reply":"2024-04-09T08:57:05.810118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.1 Aggregation with Feature Engineering\n\nAggregating electrode readings (e.g., mean, standard deviation, skewness, kurtosis) for each EEG_ID to capture general trends in brain activity associated with different patterns\n\n* **Mean**: Represents the average activity level. Besides, computing the mean electrode reading for each EEG_ID is less sensitive to outliers compared to other measures like the sum, making it a robust summary statistic\n* **Standard Deviation**: Indicates variability or dispersion around the mean, offering insights into the consistency or dynamic nature of the activity. Differences in standard deviation across EEG_IDs can indicate variability patterns in electrical activity, which may be informative for classification. For example, higher standard deviation might indicate more dynamic or diverse neural activity compared to lower standard deviation\n* **Skewness**: Measures the asymmetry of a distribution, indicating whether the data are skewed to the left (negative skew) or to the right (positive skew) relative to the mean. Positive skewness suggests that the distribution is skewed to the right, with a longer tail on the right side of the distribution. Negative skewness indicates a left-skewed distribution, with a longer tail on the left side. It may allow us to summarize the directional tendencies and asymmetry in the EEG signals\n* **Kurtosis**: Measures the \"tailedness\" of a distribution, indicating whether the data are heavy-tailed (outliers are present) or light-tailed (outliers are rare). High kurtosis implies that the distribution has heavy tails and potentially more extreme values, while low kurtosis suggests a more uniform or flat distribution. For example, spikes or sharp changes in the EEG readings may result in high kurtosis values, indicating non-normal behavior in the signal\n* **Power spectral density (PSD)**: A domain-specific feature engineering. It represents the distribution of power (or energy) of the signal across different frequencies. EEG signals are often analyzed in the frequency domain to extract frequency-related features such as power spectral density (PSD), dominant frequency, or frequency band powers (e.g., alpha, beta, theta). Hence, calculating these features from the aggregated electrode readings can provide insights into brain activity patterns","metadata":{}},{"cell_type":"code","source":"# list all the columns in the eeg dataset (sub dataset)\nprint(\"Features : \" ,eeg_sample.columns.tolist())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Aggregate Method 1** - to aggregate the electrode readings by grouping them based on time intervals (in seconds). For example, if the total row of EEG data for a single eeg_id is 11,000, the aggregation process will determine the number of bins by dividing this duration by 200 (representing a grouping interval of every 200 seconds). Subsequently, each bin will be associated with the same ID, following this pattern:\n\n* 1 to 200 seconds = Bin 1 aggregation\n* 201 to 400 seconds = Bin 2 aggregation\n* 401 to 600 seconds = Bin 3 aggregation\n* ...","metadata":{}},{"cell_type":"code","source":"\"\"\"\ndef auto_aggregate_eeg_data(df, columns_to_aggregate):\n    \n    # Step 1: read the main dataset\n    main_df = df  \n\n    # Step 2: detect EEG IDs in the main dataset and put it into an array\n    unique_eeg_ids = main_df['eeg_id'].unique()\n    \n    all_aggregated_values = []\n\n    # Step 3: iterate over detected EEG IDs\n    for eeg_id in unique_eeg_ids:\n        # Step 4: construct the file path for the sub-dataset\n        sub_dataset_path = os.path.join('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/', f\"{eeg_id}.parquet\")\n\n        # Step 5: check if the sub-dataset file exists\n        if os.path.exists(sub_dataset_path):\n            # Step 6: read the sub-dataset\n            sub_df = pd.read_parquet(sub_dataset_path)\n\n            # Step 7: aggregate values for specified columns\n            aggregated_values = {}\n            num_rows = len(sub_df)\n            sec_grouping = 200\n            \n            for i in range(num_rows // sec_grouping):\n                sec_min = i * sec_grouping + 1\n                sec_max = (i + 1) * sec_grouping\n\n                for col in columns_to_aggregate:\n                    min_value = sub_df[col][sec_min-1:sec_max].min()\n                    max_value = sub_df[col][sec_min-1:sec_max].max()\n                    mean_value = sub_df[col][sec_min-1:sec_max].mean()\n                    std_dev_value = sub_df[col][sec_min-1:sec_max].std()\n                    skewness_value = sub_df[col][sec_min-1:sec_max].skew()\n                    kurtosis_value = sub_df[col][sec_min-1:sec_max].kurt()\n                    \n                    # Step 8: to compute power spectral density (PSD) for each EEG ID\n                    f, psd = welch(sub_df[col][sec_min-1:sec_max], fs=250) # calculate power spectral density using Welch's method (Assuming EEG signal sampled at 250 Hz)\n                    total_power = np.trapz(psd, f)      # integrate PSD to get total power\n                    normalized_psd = psd / total_power  # normalize PSD by total power\n                \n                    alpha_band = np.sum(normalized_psd[(f >= 8) & (f <= 13)])  # Alpha band (8-13 Hz)\n                    beta_band = np.sum(normalized_psd[(f >= 13) & (f <= 30)])  # Beta band (13-30 Hz)\n                \n                    aggregated_values = {\n                        'eeg_id': eeg_id,\n                        'second': f\"{sec_min}-{sec_max}\",\n                        f'{col}_min': min_value,\n                        f'{col}_max': max_value,\n                        f'{col}_mean': mean_value,\n                        f'{col}_sd': std_dev_value,\n                        f'{col}_skew': skewness_value,\n                        f'{col}_kurt': kurtosis_value,\n                        f'{col}_alpha_band': alpha_band,\n                        f'{col}_beta_band': beta_band\n                    }\n\n                    all_aggregated_values.append(aggregated_values)\n\n            # Step 9: merge aggregated values back to the main dataset based on EEG ID\n            for agg_values in all_aggregated_values:\n                eeg_id = agg_values['eeg_id']\n                main_df.loc[main_df['eeg_id'] == eeg_id, agg_values.keys()] = agg_values.values()    \n                    \n    return main_df\n\n# usage:\ncolumns_to_aggregate = ['Fp1', 'F3', 'C3', 'P3', 'F7', 'T3', 'T5', 'O1', 'Fz', 'Cz', 'Pz', 'Fp2', 'F4', 'C4', 'P4', 'F8', 'T4', 'T6', 'O2'] \ndf_v2 = auto_aggregate_eeg_data(df_v1, columns_to_aggregate)\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-04-09T06:21:07.782065Z","iopub.execute_input":"2024-04-09T06:21:07.782474Z","iopub.status.idle":"2024-04-09T06:52:35.919846Z","shell.execute_reply.started":"2024-04-09T06:21:07.782443Z","shell.execute_reply":"2024-04-09T06:52:35.917978Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Aggregate Method 2** - to aggregate the electrode readings based on eeg_id","metadata":{}},{"cell_type":"code","source":"def auto_aggregate_eeg_data(df, columns_to_aggregate):\n    \n    # Step 1: read the main dataset\n    main_df = df  \n\n    # Step 2: detect EEG IDs in the main dataset and put it into an array\n    unique_eeg_ids = main_df['eeg_id'].unique()\n\n    # Step 3: iterate over detected EEG IDs\n    for eeg_id in unique_eeg_ids:\n        # Step 4: construct the file path for the sub-dataset\n        sub_dataset_path = os.path.join('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/', f\"{eeg_id}.parquet\")\n\n        # Step 5: check if the sub-dataset file exists\n        if os.path.exists(sub_dataset_path):\n            # Step 6: read the sub-dataset\n            sub_df = pd.read_parquet(sub_dataset_path)\n\n            # Step 7: aggregate values for specified columns\n            aggregated_values = {}\n            for col in columns_to_aggregate:\n                mean_value = sub_df[col].mean()\n                std_dev_value = sub_df[col].std()\n                skewness_value = sub_df[col].skew()  # skewness\n                kurtosis_value = sub_df[col].kurt()  # kurtosis\n    \n                aggregated_values[f'{col}_mean'] = mean_value\n                aggregated_values[f'{col}_sd'] = std_dev_value\n                aggregated_values[f'{col}_skew'] = skewness_value\n                aggregated_values[f'{col}_kurt'] = kurtosis_value\n                \n                # Step 8: to compute power spectral density (PSD) for each EEG ID\n                f, psd = welch(sub_df[col], fs=250) # calculate power spectral density using Welch's method (Assuming EEG signal sampled at 250 Hz)\n                total_power = np.trapz(psd, f)      # integrate PSD to get total power\n                normalized_psd = psd / total_power  # normalize PSD by total power\n                \n                alpha_band = np.sum(normalized_psd[(f >= 8) & (f <= 13)])  # Alpha band (8-13 Hz)\n                beta_band = np.sum(normalized_psd[(f >= 13) & (f <= 30)])  # Beta band (13-30 Hz)\n                \n                aggregated_values[f'{col}_alpha_band'] = alpha_band\n                aggregated_values[f'{col}_beta_band'] = beta_band\n\n            # Step 9: merge aggregated values back to the main dataset based on EEG ID\n            main_df.loc[main_df['eeg_id'] == eeg_id, aggregated_values.keys()] = aggregated_values.values()\n            \n    return main_df\n\n# usage:\ncolumns_to_aggregate = ['Fp1', 'F3', 'C3', 'P3', 'F7', 'T3', 'T5', 'O1', 'Fz', 'Cz', 'Pz', 'Fp2', 'F4', 'C4', 'P4', 'F8', 'T4', 'T6', 'O2'] \ndf_v2 = auto_aggregate_eeg_data(df_v1, columns_to_aggregate)","metadata":{"execution":{"iopub.status.busy":"2024-04-05T08:32:11.089725Z","iopub.execute_input":"2024-04-05T08:32:11.090160Z","iopub.status.idle":"2024-04-05T08:57:17.327008Z","shell.execute_reply.started":"2024-04-05T08:32:11.090124Z","shell.execute_reply":"2024-04-05T08:57:17.324725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# extract column names that end with \"_alpha_band\" as there are Fp1_alpha_band, F3_alpha_band, etc\nalpha_band_columns = [col for col in df_v2.columns if col.endswith(\"_alpha_band\")]\nbeta_band_columns = [col for col in df_v2.columns if col.endswith(\"_beta_band\")]\nmean_columns = [col for col in df_v2.columns if col.endswith(\"_mean\")]\nstd_columns = [col for col in df_v2.columns if col.endswith(\"_std\")]\nskew_columns = [col for col in df_v2.columns if col.endswith(\"_skew\")]\nkurt_columns = [col for col in df_v2.columns if col.endswith(\"_kurt\")]\n\ndf_v2[alpha_band_columns] = df_v2[alpha_band_columns].fillna(0.0)\ndf_v2[beta_band_columns] = df_v2[beta_band_columns].fillna(0.0)\ndf_v2[mean_columns] = df_v2[mean_columns].fillna(0.0)\ndf_v2[std_columns] = df_v2[std_columns].fillna(0.0)\ndf_v2[skew_columns] = df_v2[skew_columns].fillna(0.0)\ndf_v2[kurt_columns] = df_v2[kurt_columns].fillna(0.0)","metadata":{"execution":{"iopub.status.busy":"2024-04-09T09:21:56.781390Z","iopub.execute_input":"2024-04-09T09:21:56.782153Z","iopub.status.idle":"2024-04-09T09:21:56.816789Z","shell.execute_reply.started":"2024-04-09T09:21:56.782091Z","shell.execute_reply":"2024-04-09T09:21:56.814156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# to check null value again after cleanning \ndef nans(df): return df[df.isnull().any(axis=1)]\nnans(df_v2)","metadata":{"execution":{"iopub.status.busy":"2024-04-09T09:21:58.979285Z","iopub.execute_input":"2024-04-09T09:21:58.979850Z","iopub.status.idle":"2024-04-09T09:21:59.029804Z","shell.execute_reply.started":"2024-04-09T09:21:58.979808Z","shell.execute_reply":"2024-04-09T09:21:59.028519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_columns', None)\ndf_v2[df_v2['eeg_id'] == 2277392603]","metadata":{"execution":{"iopub.status.busy":"2024-04-09T08:06:42.527182Z","iopub.execute_input":"2024-04-09T08:06:42.528575Z","iopub.status.idle":"2024-04-09T08:06:42.655317Z","shell.execute_reply.started":"2024-04-09T08:06:42.528523Z","shell.execute_reply":"2024-04-09T08:06:42.654045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.0 Correlation","metadata":{}},{"cell_type":"code","source":"print(\"Features : \" ,df_v2.columns.tolist())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# drop meaningless columns\nobv_df = df_v2.drop(columns = ['eeg_id', 'spectrogram_id'])\n# remain few columns to observe their relationship\nobv_df = obv_df.loc[:, ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote', 'Fp1_mean', 'Fp1_sd', 'Fp1_skew', 'Fp1_kurt', 'Fp1_alpha_band', 'Fp1_beta_band', 'F3_mean','F3_sd', 'F3_skew', 'F3_kurt', 'F3_alpha_band', 'F3_beta_band', 'C3_mean', 'C3_sd', 'C3_skew', 'C3_kurt', 'C3_alpha_band', 'C3_beta_band']]\n\nplt.figure(figsize = (18, 10))\nsns.heatmap(obv_df.corr(), cmap=\"Blues\", annot = True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-05T09:09:25.230798Z","iopub.execute_input":"2024-04-05T09:09:25.231334Z","iopub.status.idle":"2024-04-05T09:09:27.972783Z","shell.execute_reply.started":"2024-04-05T09:09:25.231294Z","shell.execute_reply":"2024-04-05T09:09:27.970187Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3.0 Split Dataset\n\nTo compute a hash of each instance's identifier and put that instance in the train and test set so as to ensure that both train and test set will always remain consistent across multiple runs","metadata":{}},{"cell_type":"code","source":"def test_set_check(identifier, test_ratio):\n    return crc32(np.int64(identifier)) & 0xffffffff < test_ratio * 2**32\n\ndef split_train_test_by_id(data, test_ratio, id_column):\n    ids = data[id_column]\n    in_test_set = ids.apply(lambda id_: test_set_check(id_, test_ratio))\n    return data.loc[~in_test_set], data.loc[in_test_set]","metadata":{"execution":{"iopub.status.busy":"2024-04-05T09:09:34.207790Z","iopub.execute_input":"2024-04-05T09:09:34.208312Z","iopub.status.idle":"2024-04-05T09:09:34.217640Z","shell.execute_reply.started":"2024-04-05T09:09:34.208269Z","shell.execute_reply":"2024-04-05T09:09:34.215787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# drop string column \ndf_v2_train = df_v2.drop(['expert_consensus'], axis=1) \n\n# add index column\ndf_with_id = df_v2_train.reset_index()\ntrain_set, test_set = split_train_test_by_id(df_with_id, 0.2, \"index\") \n\n# to check whether will get random row whenever rerun\n#column_values= test_set['index'].tolist() \n\n# view dataset\nprint('\\033[1m' + \"train_set:\" + '\\033[0m')\ndisplay(train_set.head(3))\n#print('\\033[1m' + \"test_set1:\" + '\\033[0m')\n#display(test_set1.head(3))","metadata":{"execution":{"iopub.status.busy":"2024-04-05T09:09:35.130098Z","iopub.execute_input":"2024-04-05T09:09:35.130620Z","iopub.status.idle":"2024-04-05T09:09:35.289976Z","shell.execute_reply.started":"2024-04-05T09:09:35.130581Z","shell.execute_reply":"2024-04-05T09:09:35.288786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# independent variable (iv) / features\nX = train_set.drop(['index', 'seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote'], axis=1)\nx = test_set.drop(['index', 'seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote'], axis=1)\n\n# dependent variable (dv) / target  \nY = train_set[['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']]\ny = test_set[['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']]","metadata":{"execution":{"iopub.status.busy":"2024-04-05T09:09:36.820095Z","iopub.execute_input":"2024-04-05T09:09:36.820548Z","iopub.status.idle":"2024-04-05T09:09:36.845383Z","shell.execute_reply.started":"2024-04-05T09:09:36.820514Z","shell.execute_reply":"2024-04-05T09:09:36.844101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# select target variables\ntarget_variables = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']","metadata":{"execution":{"iopub.status.busy":"2024-04-05T09:09:37.817864Z","iopub.execute_input":"2024-04-05T09:09:37.818417Z","iopub.status.idle":"2024-04-05T09:09:37.825081Z","shell.execute_reply.started":"2024-04-05T09:09:37.818373Z","shell.execute_reply":"2024-04-05T09:09:37.823502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4.0 Normalization","metadata":{}},{"cell_type":"code","source":"# set up the figure and axes\nfig, axes = plt.subplots(nrows=len(target_variables), ncols=1, figsize=(10, 8))\nfig.suptitle('Distribution of Target Variables', y=1.02)\n\nfor i, target in enumerate(target_variables):\n    # plot histogram with kernel density estimate\n    sns.histplot(df_v2[target], kde=True, ax=axes[i], bins=30)\n\n    # Add mean and median lines\n    mean_value = df_v2[target].mean()\n    median_value = df_v2[target].median()\n\n    axes[i].axvline(mean_value, color='red', linestyle='dashed', linewidth=2, label=f'Mean: {mean_value:.2f}')\n    axes[i].axvline(median_value, color='green', linestyle='dashed', linewidth=2, label=f'Median: {median_value:.2f}')\n\n    # Set labels and title\n    axes[i].set_xlabel(target)\n    axes[i].set_ylabel('Frequency')\n    axes[i].legend()\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Given that your target variables have varying means and medians, using RobustScaler might be a prudent choice. This can help ensure that the scaling process is less influenced by extreme values.","metadata":{}},{"cell_type":"code","source":"# normalization is needed if there is multiple iv\nscaler = RobustScaler()\nX = scaler.fit_transform(X)\nx = scaler.transform(x)\n\nprint(\"Train dv: \", X.shape, \"iv: \", Y.shape)\nprint(\"Test dv: \", x.shape, \"iv: \", y.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-05T09:19:47.386842Z","iopub.execute_input":"2024-04-05T09:19:47.387983Z","iopub.status.idle":"2024-04-05T09:19:47.546633Z","shell.execute_reply.started":"2024-04-05T09:19:47.387930Z","shell.execute_reply":"2024-04-05T09:19:47.545703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5.0 Model Training\n### 5.1 Linear Regression ML Model","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\n\n# to store trained models and predictions\nlr_models = {}\n\n# initialize separate LinearRegression for each target variable\nlr = {target: LinearRegression() for target in target_variables}\n\n# train each classifier\nfor target in target_variables:\n    lr[target].fit(X, Y[target])\n    lr_models[target] = lr[target] \n\n# predict on the test set\nlr_y_pred = {target: lr[target].predict(x) for target in target_variables}\n\n# evaluate the model\nfor target in target_variables:\n    mse = mean_squared_error(y[target], lr_y_pred[target])\n    print('\\033[1m' + f'MSE for {target}: {mse:.2f}' + '\\033[0m')\n    mae = mean_absolute_error(y[target], lr_y_pred[target])\n    print('\\033[1m' + f'MAE for {target}: {mae:.2f}' + '\\033[0m')\n    r2 = r2_score(y[target], lr_y_pred[target])\n    print('\\033[1m' + f'R2 for {target}: {r2:.2f}' + '\\033[0m')\n    print()","metadata":{"execution":{"iopub.status.busy":"2024-04-05T09:19:51.206524Z","iopub.execute_input":"2024-04-05T09:19:51.207000Z","iopub.status.idle":"2024-04-05T09:19:52.412234Z","shell.execute_reply.started":"2024-04-05T09:19:51.206957Z","shell.execute_reply":"2024-04-05T09:19:52.410600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 5.2 Decision Tree Regression ML Model","metadata":{}},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeRegressor\n\ndt_models = {}\ndt = {target: DecisionTreeRegressor(random_state=42) for target in target_variables}\n\n# train each classifier\nfor target in target_variables:\n    dt[target].fit(X, Y[target])\n    dt_models[target] = dt[target] \n\n# predict on the test set\ndt_y_pred = {target: dt[target].predict(x) for target in target_variables}\n\n# evaluate the model\nfor target in target_variables:\n    mse = mean_squared_error(y[target], lr_y_pred[target])\n    print('\\033[1m' + f'MSE for {target}: {mse:.2f}' + '\\033[0m')\n    mae = mean_absolute_error(y[target], lr_y_pred[target])\n    print('\\033[1m' + f'MAE for {target}: {mae:.2f}' + '\\033[0m')\n    r2 = r2_score(y[target], lr_y_pred[target])\n    print('\\033[1m' + f'R2 for {target}: {r2:.2f}' + '\\033[0m')\n    print()","metadata":{"execution":{"iopub.status.busy":"2024-04-05T09:19:54.354525Z","iopub.execute_input":"2024-04-05T09:19:54.356264Z","iopub.status.idle":"2024-04-05T09:20:20.075182Z","shell.execute_reply.started":"2024-04-05T09:19:54.356211Z","shell.execute_reply":"2024-04-05T09:20:20.073994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 5.3 Random Forest Regressor ML Model","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\n\nrf_models = {}\nrf = {target: RandomForestRegressor(random_state=42) for target in target_variables}\n\n# train each classifier\nfor target in target_variables:\n    rf[target].fit(X, Y[target])\n    rf_models[target] = rf[target] \n\n# predict on the test set\nrf_y_pred = {target: rf[target].predict(x) for target in target_variables}\n\n# evaluate the model\nfor target in target_variables:\n    mse = mean_squared_error(y[target], lr_y_pred[target])\n    print('\\033[1m' + f'MSE for {target}: {mse:.2f}' + '\\033[0m')\n    mae = mean_absolute_error(y[target], lr_y_pred[target])\n    print('\\033[1m' + f'MAE for {target}: {mae:.2f}' + '\\033[0m')\n    r2 = r2_score(y[target], lr_y_pred[target])\n    print('\\033[1m' + f'R2 for {target}: {r2:.2f}' + '\\033[0m')\n    print()","metadata":{"execution":{"iopub.status.busy":"2024-04-05T09:20:20.077435Z","iopub.execute_input":"2024-04-05T09:20:20.077819Z","iopub.status.idle":"2024-04-05T09:49:12.362931Z","shell.execute_reply.started":"2024-04-05T09:20:20.077785Z","shell.execute_reply":"2024-04-05T09:49:12.361467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 5.4 Support Vector Regression ML Model","metadata":{}},{"cell_type":"code","source":"from sklearn.svm import SVR\n\nsvr_models = {}\nsvr = {target: SVR() for target in target_variables}\n\n# train each classifier\nfor target in target_variables:\n    svr[target].fit(X, Y[target])\n    svr_models[target] = svr[target] \n\n# predict on the test set\nsvr_y_pred = {target: svr[target].predict(x) for target in target_variables}\n\n# evaluate the model\nfor target in target_variables:\n    mse = mean_squared_error(y[target], svr_y_pred[target])\n    print('\\033[1m' + f'MSE for {target}: {mse:.2f}' + '\\033[0m')\n    mae = mean_absolute_error(y[target], svr_y_pred[target])\n    print('\\033[1m' + f'MAE for {target}: {mae:.2f}' + '\\033[0m')\n    r2 = r2_score(y[target], svr_y_pred[target])\n    print('\\033[1m' + f'R2 for {target}: {r2:.2f}' + '\\033[0m')\n    print()","metadata":{"execution":{"iopub.status.busy":"2024-04-05T09:49:12.364766Z","iopub.execute_input":"2024-04-05T09:49:12.365260Z","iopub.status.idle":"2024-04-05T09:52:49.695471Z","shell.execute_reply.started":"2024-04-05T09:49:12.365213Z","shell.execute_reply":"2024-04-05T09:52:49.694231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 5.5 Gradient Boosting Regression ML Model","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingRegressor\n\ngb_models = {}\ngb = {target: GradientBoostingRegressor(random_state=42) for target in target_variables}\n\n# train each classifier\nfor target in target_variables:\n    gb[target].fit(X, Y[target])\n    gb_models[target] = gb[target] \n\n# predict on the test set\ngb_y_pred = {target: gb[target].predict(x) for target in target_variables}\n\n# evaluate the model\nfor target in target_variables:\n    mse = mean_squared_error(y[target], gb_y_pred[target])\n    print('\\033[1m' + f'MSE for {target}: {mse:.2f}' + '\\033[0m')\n    mae = mean_absolute_error(y[target], gb_y_pred[target])\n    print('\\033[1m' + f'MAE for {target}: {mae:.2f}' + '\\033[0m')\n    r2 = r2_score(y[target], gb_y_pred[target])\n    print('\\033[1m' + f'R2 for {target}: {r2:.2f}' + '\\033[0m')\n    print()","metadata":{"execution":{"iopub.status.busy":"2024-04-05T09:52:49.697909Z","iopub.execute_input":"2024-04-05T09:52:49.698387Z","iopub.status.idle":"2024-04-05T10:00:15.909803Z","shell.execute_reply.started":"2024-04-05T09:52:49.698352Z","shell.execute_reply":"2024-04-05T10:00:15.908217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 5.6 Ensemble ML Model","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import VotingRegressor\n\nensemble_models = {}\nfor target in target_variables:\n    # create a voting ensemble model\n    ensemble_model = VotingRegressor([('Linear Regression', lr_models[target]),\n                                      ('Decision Tree', dt_models[target]),\n                                      ('Random Forest', rf_models[target]),\n                                      ('SVR' ,svr_models[target]),\n                                      ('Gradient Boosting', gb_models[target])])\n    \n    # fit the individual models\n    ensemble_model.fit(X, Y[target])\n    ensemble_models[target] = ensemble_model\n    \n# make predictions using the ensemble models\nensemble_predictions = {target: model.predict(x) for target, model in ensemble_models.items()}\n\nfor target in target_variables:\n    mse = mean_squared_error(y[target], ensemble_predictions[target])\n    print('\\033[1m' + f'MSE for {target}: {mse:.2f}' + '\\033[0m')\n    mae = mean_absolute_error(y[target], ensemble_predictions[target])\n    print('\\033[1m' + f'MAE for {target}: {mae:.2f}' + '\\033[0m')\n    r2 = r2_score(y[target], ensemble_predictions[target])\n    print('\\033[1m' + f'R2 for {target}: {r2:.2f}' + '\\033[0m')\n    print()","metadata":{"execution":{"iopub.status.busy":"2024-04-05T10:00:15.911892Z","iopub.execute_input":"2024-04-05T10:00:15.912518Z","iopub.status.idle":"2024-04-05T10:40:53.795195Z","shell.execute_reply.started":"2024-04-05T10:00:15.912467Z","shell.execute_reply":"2024-04-05T10:40:53.793875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 6.0 Test","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\ntest_df","metadata":{"execution":{"iopub.status.busy":"2024-04-05T10:41:43.591181Z","iopub.execute_input":"2024-04-05T10:41:43.591652Z","iopub.status.idle":"2024-04-05T10:41:43.611282Z","shell.execute_reply.started":"2024-04-05T10:41:43.591612Z","shell.execute_reply":"2024-04-05T10:41:43.609603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sub dataset\ntest_eeg = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/3911565283.parquet')\ntest_eeg.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-03-31T12:04:13.987526Z","iopub.execute_input":"2024-03-31T12:04:13.988200Z","iopub.status.idle":"2024-03-31T12:04:14.021330Z","shell.execute_reply.started":"2024-03-31T12:04:13.988153Z","shell.execute_reply":"2024-03-31T12:04:14.020404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def auto_aggregate_eeg_data(df, columns_to_aggregate):\n    \n    # Step 1: read the main dataset\n    main_df = df  \n\n    # Step 2: detect EEG IDs in the main dataset and put it into an array\n    unique_eeg_ids = main_df['eeg_id'].unique()\n\n    # Step 3: iterate over detected EEG IDs\n    for eeg_id in unique_eeg_ids:\n        # Step 4: construct the file path for the sub-dataset\n        sub_dataset_path = os.path.join('/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/', f\"{eeg_id}.parquet\")\n\n        # Step 5: check if the sub-dataset file exists\n        if os.path.exists(sub_dataset_path):\n            # Step 6: read the sub-dataset\n            sub_df = pd.read_parquet(sub_dataset_path)\n\n            # Step 7: aggregate values for specified columns\n            aggregated_values = {}\n            for col in columns_to_aggregate:\n                mean_value = sub_df[col].mean()\n                std_dev_value = sub_df[col].std()\n                skewness_value = sub_df[col].skew()  # skewness\n                kurtosis_value = sub_df[col].kurt()  # kurtosis\n    \n                aggregated_values[f'{col}_mean'] = mean_value\n                aggregated_values[f'{col}_sd'] = std_dev_value\n                aggregated_values[f'{col}_skew'] = skewness_value\n                aggregated_values[f'{col}_kurt'] = kurtosis_value\n                \n                # Step 8: to compute power spectral density (PSD) for each EEG ID\n                f, psd = welch(sub_df[col], fs=250) # calculate power spectral density using Welch's method (Assuming EEG signal sampled at 250 Hz)\n                total_power = np.trapz(psd, f)      # integrate PSD to get total power\n                normalized_psd = psd / total_power  # normalize PSD by total power\n                \n                alpha_band = np.sum(normalized_psd[(f >= 8) & (f <= 13)])  # Alpha band (8-13 Hz)\n                beta_band = np.sum(normalized_psd[(f >= 13) & (f <= 30)])  # Beta band (13-30 Hz)\n                \n                aggregated_values[f'{col}_alpha_band'] = alpha_band\n                aggregated_values[f'{col}_beta_band'] = beta_band\n\n            # Step 9: merge aggregated values back to the main dataset based on EEG ID\n            main_df.loc[main_df['eeg_id'] == eeg_id, aggregated_values.keys()] = aggregated_values.values()\n            \n    return main_df\n\n# usage:\ncolumns_to_aggregate = ['Fp1', 'F3', 'C3', 'P3', 'F7', 'T3', 'T5', 'O1', 'Fz', 'Cz', 'Pz', 'Fp2', 'F4', 'C4', 'P4', 'F8', 'T4', 'T6', 'O2'] \ntest_df_v2 = auto_aggregate_eeg_data(test_df, columns_to_aggregate)","metadata":{"execution":{"iopub.status.busy":"2024-04-05T10:41:47.740832Z","iopub.execute_input":"2024-04-05T10:41:47.741271Z","iopub.status.idle":"2024-04-05T10:41:47.854143Z","shell.execute_reply.started":"2024-04-05T10:41:47.741236Z","shell.execute_reply":"2024-04-05T10:41:47.852771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"alpha_band_columns = [col for col in test_df_v2.columns if col.endswith(\"_alpha_band\")]\nbeta_band_columns = [col for col in test_df_v2.columns if col.endswith(\"_beta_band\")]\nmean_columns = [col for col in test_df_v2.columns if col.endswith(\"_mean\")]\nstd_columns = [col for col in test_df_v2.columns if col.endswith(\"_std\")]\nskew_columns = [col for col in test_df_v2.columns if col.endswith(\"_skew\")]\nkurt_columns = [col for col in test_df_v2.columns if col.endswith(\"_kurt\")]\n\ntest_df_v2[alpha_band_columns] = test_df_v2[alpha_band_columns].fillna(0.0)\ntest_df_v2[beta_band_columns] = test_df_v2[beta_band_columns].fillna(0.0)\ntest_df_v2[mean_columns] = test_df_v2[mean_columns].fillna(0.0)\ntest_df_v2[std_columns] = test_df_v2[std_columns].fillna(0.0)\ntest_df_v2[skew_columns] = test_df_v2[skew_columns].fillna(0.0)\ntest_df_v2[kurt_columns] = test_df_v2[kurt_columns].fillna(0.0)","metadata":{"execution":{"iopub.status.busy":"2024-04-01T04:44:30.731312Z","iopub.execute_input":"2024-04-01T04:44:30.732108Z","iopub.status.idle":"2024-04-01T04:44:30.782144Z","shell.execute_reply.started":"2024-04-01T04:44:30.732064Z","shell.execute_reply":"2024-04-01T04:44:30.780790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df_v2\n#train_for_test = df_v2_train.drop(columns = target_variables) # for testing","metadata":{"execution":{"iopub.status.busy":"2024-04-05T10:47:31.370815Z","iopub.execute_input":"2024-04-05T10:47:31.371827Z","iopub.status.idle":"2024-04-05T10:47:31.397995Z","shell.execute_reply.started":"2024-04-05T10:47:31.371785Z","shell.execute_reply":"2024-04-05T10:47:31.396565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ensemble_predictions_new = {}\nfor target, model in ensemble_models.items():\n    #predictions = model.predict(train_for_test)\n    predictions = model.predict(test_df_v2)\n    ensemble_predictions_new[target] = predictions","metadata":{"execution":{"iopub.status.busy":"2024-04-05T10:47:35.963639Z","iopub.execute_input":"2024-04-05T10:47:35.964181Z","iopub.status.idle":"2024-04-05T10:47:36.138201Z","shell.execute_reply.started":"2024-04-05T10:47:35.964138Z","shell.execute_reply":"2024-04-05T10:47:36.136618Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# to obtain min, max, sum values based on each row \n# get the number of indices (assuming all keys have the same length)\nnum_indices = len(next(iter(ensemble_predictions_new.values())))\nmin_values = np.empty(num_indices)\nmax_values = np.empty(num_indices)\nsum_values = np.empty(num_indices)\n\n# iterate over each index (to find min and max values)\nfor i in range(num_indices):\n    \n    # extract values at index i for all keys into a numpy array\n    values_at_index = np.array([arr[i] for arr in ensemble_predictions_new.values()])\n    \n    # find the min and max values at index i across all keys\n    min_value_at_index = np.min(values_at_index)\n    max_value_at_index = np.max(values_at_index)\n    sum_value_at_index = np.sum(abs(values_at_index)) # convert neg to post value\n    # store the values\n    min_values[i] = min_value_at_index\n    max_values[i] = max_value_at_index\n    sum_values[i] = sum_value_at_index\n\nprint(\"min val = \", min_values)\nprint(\"max val = \", max_values)\nprint(\"sum val = \", sum_values)","metadata":{"execution":{"iopub.status.busy":"2024-04-05T10:47:42.444851Z","iopub.execute_input":"2024-04-05T10:47:42.445347Z","iopub.status.idle":"2024-04-05T10:47:42.457187Z","shell.execute_reply.started":"2024-04-05T10:47:42.445308Z","shell.execute_reply":"2024-04-05T10:47:42.455699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"modified_target_val = {}\nvote_target_val = {}\n\nfor target in target_variables:\n    \n    values = ensemble_predictions_new[target]\n    # to store the modified values for the current target variable\n    modified_values = []\n    vote_values = []\n    \n    # iterate over each value and its corresponding index\n    for value, sum_val in zip(values, sum_values):\n        modified_value = abs(value) / sum_val\n        vote_value = modified_value * 45\n        vote_value = round((round(vote_value) / 45), 1)\n        vote_values.append(vote_value)\n    \n    # convert the list of modified values to a numpy array and store it in the dictionary\n    vote_target_val[target] = np.array(vote_values)\n\nprint(vote_target_val)","metadata":{"execution":{"iopub.status.busy":"2024-04-05T10:47:43.369820Z","iopub.execute_input":"2024-04-05T10:47:43.370222Z","iopub.status.idle":"2024-04-05T10:47:43.380979Z","shell.execute_reply.started":"2024-04-05T10:47:43.370191Z","shell.execute_reply":"2024-04-05T10:47:43.379458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sum_rows = np.empty(num_indices)\ndiff = np.empty(num_indices)\n\n# iterate over each index (to find min and max values)\nfor i in range(num_indices):\n    \n    # extract values at index i for all keys into a numpy array\n    values_at_index = np.array([arr[i] for arr in vote_target_val.values()])\n        \n    # find the min and max values at index i across all keys\n    sum_row = np.sum(values_at_index)\n    # store the values\n    sum_rows[i] = sum_row\n    diff[i] = 1.0 - sum_rows[i]\n    \n    # Find the target column(s) with the highest value\n    max_column_indices = np.where(values_at_index == np.max(values_at_index))[0]\n    max_columns = [list(vote_target_val.keys())[idx] for idx in max_column_indices]\n    \n    # randomly select one of the target columns with the highest value\n    selected_column = np.random.choice(max_columns)\n    \n    # add the difference to the selected target column\n    vote_target_val[selected_column][i] += diff[i]","metadata":{"execution":{"iopub.status.busy":"2024-04-05T10:47:46.270983Z","iopub.execute_input":"2024-04-05T10:47:46.271450Z","iopub.status.idle":"2024-04-05T10:47:46.282283Z","shell.execute_reply.started":"2024-04-05T10:47:46.271416Z","shell.execute_reply":"2024-04-05T10:47:46.280925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create a DataFrame for submission\n#submission = pd.DataFrame(columns=['eeg_id'] + list(final_target_val.keys()))\nsubmission = pd.DataFrame(columns=['eeg_id'] + list(vote_target_val.keys()))\n\n# assign eeg_id column from test_df_v2 to submission \nsubmission['eeg_id'] = test_df_v2['eeg_id']\n#submission['eeg_id'] = train_for_test['eeg_id'] # test\n\n# sssign predicted target values to submission \nfor target in vote_target_val:\n    submission[target] = vote_target_val[target]\n    \nsubmission.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-04-05T10:48:00.054572Z","iopub.execute_input":"2024-04-05T10:48:00.055634Z","iopub.status.idle":"2024-04-05T10:48:00.081850Z","shell.execute_reply.started":"2024-04-05T10:48:00.055558Z","shell.execute_reply":"2024-04-05T10:48:00.080476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.fillna(0.0, inplace=True)\nsubmission.to_csv('submission.csv', index=False) \nsubmission.dtypes","metadata":{"execution":{"iopub.status.busy":"2024-04-05T10:48:04.091721Z","iopub.execute_input":"2024-04-05T10:48:04.092202Z","iopub.status.idle":"2024-04-05T10:48:04.107113Z","shell.execute_reply.started":"2024-04-05T10:48:04.092167Z","shell.execute_reply":"2024-04-05T10:48:04.105742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dff = pd.read_csv('/kaggle/working/submission.csv')\ndff.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-04-05T10:48:05.504800Z","iopub.execute_input":"2024-04-05T10:48:05.505279Z","iopub.status.idle":"2024-04-05T10:48:05.525886Z","shell.execute_reply.started":"2024-04-05T10:48:05.505244Z","shell.execute_reply":"2024-04-05T10:48:05.523934Z"},"trusted":true},"execution_count":null,"outputs":[]}]}