{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\n\nimport os\nimport shutil","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-29T21:51:17.017254Z","iopub.execute_input":"2024-03-29T21:51:17.018096Z","iopub.status.idle":"2024-03-29T21:51:17.025406Z","shell.execute_reply.started":"2024-03-29T21:51:17.018047Z","shell.execute_reply":"2024-03-29T21:51:17.024148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_palette('copper')","metadata":{"execution":{"iopub.status.busy":"2024-03-29T21:51:17.027805Z","iopub.execute_input":"2024-03-29T21:51:17.028526Z","iopub.status.idle":"2024-03-29T21:51:17.038254Z","shell.execute_reply.started":"2024-03-29T21:51:17.028492Z","shell.execute_reply":"2024-03-29T21:51:17.037153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_folder(directory_path):\n    # Check if the directory exists\n    if os.path.exists(directory_path):\n        # Remove the directory and all its contents\n        shutil.rmtree(directory_path)\n\n    # Create the directory\n    os.makedirs(directory_path)","metadata":{"execution":{"iopub.status.busy":"2024-03-29T21:51:17.040139Z","iopub.execute_input":"2024-03-29T21:51:17.040913Z","iopub.status.idle":"2024-03-29T21:51:17.050105Z","shell.execute_reply.started":"2024-03-29T21:51:17.040879Z","shell.execute_reply":"2024-03-29T21:51:17.049022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_idealized_cases(remove_overlapping_samples):\n    data = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\n\n    target_columns = [\n        'seizure_vote',\n        'lpd_vote',\n        'gpd_vote',\n        'lrda_vote',\n        'grda_vote',\n        'other_vote'\n    ]\n    data['total_votes'] = data[target_columns].sum(axis=1)\n    data[target_columns] = data[target_columns].div(data[target_columns].sum(axis=1), axis=0)\n    data['first_max_agreement'] = data[target_columns].max(axis=1)\n\n    temp = data[target_columns].to_numpy()\n    temp = np.sort(temp, axis=1)\n    data['second_max_agreement'] = temp[:, 4]\n\n    #remove overlapping samples\n    if remove_overlapping_samples:\n        data = data.groupby('eeg_id').agg({\n            'eeg_label_offset_seconds':'first',\n            'spectrogram_id':'first',\n            'spectrogram_sub_id':'first',\n            'spectrogram_label_offset_seconds':'first',\n            'label_id':'first',\n            'patient_id':'first',\n            'expert_consensus': 'first',\n            'seizure_vote': 'first',\n            'lpd_vote':'first',\n            'gpd_vote':'first',\n            'lrda_vote': 'first',\n            'grda_vote': 'first',\n            'other_vote':'first',\n            'total_votes':'first',\n            'first_max_agreement': 'first',\n            'second_max_agreement': 'first'\n        }).reset_index()\n\n    data = data.loc[\n        ( data['first_max_agreement'] >= 0.5 ) &\n        ( abs(data['first_max_agreement'] - data['second_max_agreement']) >= 0.2 )\n    ]\n\n    eeg_unique_consensus = data.groupby(['eeg_id', 'patient_id'])['expert_consensus'].nunique().reset_index()\n    eeg_unique_consensus = eeg_unique_consensus.loc[eeg_unique_consensus['expert_consensus'] == 1]\n\n    train_idx, test_idx = train_test_split(eeg_unique_consensus['patient_id'].unique(), test_size=0.2)\n\n    train_data = data.loc[data['patient_id'].isin(train_idx)]\n    test_data = data.loc[data['patient_id'].isin(test_idx)]\n    \n    return train_data, test_data","metadata":{"execution":{"iopub.status.busy":"2024-03-29T21:51:17.051676Z","iopub.execute_input":"2024-03-29T21:51:17.052363Z","iopub.status.idle":"2024-03-29T21:51:17.065877Z","shell.execute_reply.started":"2024-03-29T21:51:17.052309Z","shell.execute_reply":"2024-03-29T21:51:17.064611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_proto_cases(remove_overlapping_samples):\n    data = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\n\n    target_columns = [\n        'seizure_vote',\n        'lpd_vote',\n        'gpd_vote',\n        'lrda_vote',\n        'grda_vote',\n        'other_vote'\n    ]\n    data['total_votes'] = data[target_columns].sum(axis=1)\n    data[target_columns] = data[target_columns].div(data[target_columns].sum(axis=1), axis=0)\n    data['first_max_agreement'] = data[target_columns].max(axis=1)\n\n    temp = data[target_columns].to_numpy()\n    temp = np.sort(temp, axis=1)\n    data['second_max_agreement'] = temp[:, 4]\n\n    #remove overlapping samples\n    if remove_overlapping_samples:\n        data = data.groupby('eeg_id').agg({\n            'eeg_label_offset_seconds':'first',\n            'spectrogram_id':'first',\n            'spectrogram_sub_id':'first',\n            'spectrogram_label_offset_seconds':'first',\n            'label_id':'first',\n            'patient_id':'first',\n            'expert_consensus': 'first',\n            'seizure_vote': 'first',\n            'lpd_vote':'first',\n            'gpd_vote':'first',\n            'lrda_vote': 'first',\n            'grda_vote': 'first',\n            'other_vote':'first',\n            'total_votes':'first',\n            'first_max_agreement': 'first',\n            'second_max_agreement': 'first'\n        }).reset_index()\n\n    data = data.loc[\n        (data['first_max_agreement'] + data['second_max_agreement'] >= 0.8) &\n        ( (data['other_vote'] == data['first_max_agreement']) | (data['other_vote'] == data['second_max_agreement'])  ) &\n        ( abs(data['first_max_agreement'] - data['second_max_agreement']) <= 0.1 )\n    ]\n\n    eeg_unique_consensus = data.groupby(['eeg_id', 'patient_id'])['expert_consensus'].nunique().reset_index()\n    eeg_unique_consensus = eeg_unique_consensus.loc[eeg_unique_consensus['expert_consensus'] == 1]\n\n    train_idx, test_idx = train_test_split(eeg_unique_consensus['patient_id'].unique(), test_size=0.2)\n\n    train_data = data.loc[data['patient_id'].isin(train_idx)]\n    test_data = data.loc[data['patient_id'].isin(test_idx)]\n    \n    return train_data, test_data","metadata":{"execution":{"iopub.status.busy":"2024-03-29T21:51:17.068831Z","iopub.execute_input":"2024-03-29T21:51:17.069781Z","iopub.status.idle":"2024-03-29T21:51:17.084370Z","shell.execute_reply.started":"2024-03-29T21:51:17.069741Z","shell.execute_reply":"2024-03-29T21:51:17.083149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_edge_cases(remove_overlapping_samples):\n    data = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\n\n    target_columns = [\n        'seizure_vote',\n        'lpd_vote',\n        'gpd_vote',\n        'lrda_vote',\n        'grda_vote',\n        'other_vote'\n    ]\n    data['total_votes'] = data[target_columns].sum(axis=1)\n    data[target_columns] = data[target_columns].div(data[target_columns].sum(axis=1), axis=0)\n    data['first_max_agreement'] = data[target_columns].max(axis=1)\n\n    temp = data[target_columns].to_numpy()\n    temp = np.sort(temp, axis=1)\n    data['second_max_agreement'] = temp[:, 4]\n\n    #remove overlapping samples\n    if remove_overlapping_samples:\n        data = data.groupby('eeg_id').agg({\n            'eeg_label_offset_seconds':'first',\n            'spectrogram_id':'first',\n            'spectrogram_sub_id':'first',\n            'spectrogram_label_offset_seconds':'first',\n            'label_id':'first',\n            'patient_id':'first',\n            'expert_consensus': 'first',\n            'seizure_vote': 'first',\n            'lpd_vote':'first',\n            'gpd_vote':'first',\n            'lrda_vote': 'first',\n            'grda_vote': 'first',\n            'other_vote':'first',\n            'total_votes':'first',\n            'first_max_agreement': 'first',\n            'second_max_agreement': 'first'\n        }).reset_index()\n\n    data = data.loc[\n        (data['first_max_agreement'] + data['second_max_agreement'] >= 0.8) &\n        ( (data['other_vote'] != data['first_max_agreement']) & (data['other_vote'] != data['second_max_agreement'])  ) &\n        ( abs(data['first_max_agreement'] - data['second_max_agreement']) <= 0.1 )\n    ]\n\n    eeg_unique_consensus = data.groupby(['eeg_id', 'patient_id'])['expert_consensus'].nunique().reset_index()\n    eeg_unique_consensus = eeg_unique_consensus.loc[eeg_unique_consensus['expert_consensus'] == 1]\n\n    train_idx, test_idx = train_test_split(eeg_unique_consensus['patient_id'].unique(), test_size=0.2)\n\n    train_data = data.loc[data['patient_id'].isin(train_idx)]\n    test_data = data.loc[data['patient_id'].isin(test_idx)]\n    \n    return train_data, test_data","metadata":{"execution":{"iopub.status.busy":"2024-03-29T21:51:17.114186Z","iopub.execute_input":"2024-03-29T21:51:17.114610Z","iopub.status.idle":"2024-03-29T21:51:17.128470Z","shell.execute_reply.started":"2024-03-29T21:51:17.114575Z","shell.execute_reply":"2024-03-29T21:51:17.127295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The EEG segments used in this competition have been annotated, or classified, by a group of experts. In some cases experts completely agree about the correct label. On other cases the experts disagree. We call segments where there are high levels of agreement “idealized” patterns. Cases where ~1/2 of experts give a label as “other” and ~1/2 give one of the remaining five labels, we call “proto patterns”. Cases where experts are approximately split between 2 of the 5 named patterns, we call “edge cases”.","metadata":{}},{"cell_type":"markdown","source":"# Idealized Cases","metadata":{}},{"cell_type":"markdown","source":"## With Overlapping Samples","metadata":{}},{"cell_type":"code","source":"remove_overlapping_samples = False\n\ntrain_data, test_data = extract_idealized_cases(remove_overlapping_samples)\n\nassert len( train_data.loc[train_data['eeg_id'].isin(test_data['eeg_id'].to_list())] ) == 0  # same eeg_id cannot be in train and test samples\nassert len( train_data.loc[train_data['patient_id'].isin(test_data['patient_id'].to_list())] ) == 0 # same patient_id cannot be in train and test samples","metadata":{"execution":{"iopub.status.busy":"2024-03-29T21:51:17.130389Z","iopub.execute_input":"2024-03-29T21:51:17.130806Z","iopub.status.idle":"2024-03-29T21:51:17.786022Z","shell.execute_reply.started":"2024-03-29T21:51:17.130734Z","shell.execute_reply":"2024-03-29T21:51:17.784034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots()\nsns.countplot(train_data, x='expert_consensus', ax=ax)\nax.set_title('Count of expert consensus')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-29T21:51:17.788328Z","iopub.execute_input":"2024-03-29T21:51:17.788819Z","iopub.status.idle":"2024-03-29T21:51:18.333980Z","shell.execute_reply.started":"2024-03-29T21:51:17.788773Z","shell.execute_reply":"2024-03-29T21:51:18.332766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"create_folder('/kaggle/working/idealized/with_overlapping')\n\ntrain_data.to_csv('/kaggle/working/idealized/with_overlapping/train.csv', index=False)\ntest_data.to_csv('/kaggle/working/idealized/with_overlapping/test.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-03-29T21:51:18.337144Z","iopub.execute_input":"2024-03-29T21:51:18.337926Z","iopub.status.idle":"2024-03-29T21:51:19.958738Z","shell.execute_reply.started":"2024-03-29T21:51:18.337881Z","shell.execute_reply":"2024-03-29T21:51:19.957463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Without Overlapping Samples","metadata":{}},{"cell_type":"code","source":"remove_overlapping_samples = True\n\ntrain_data, test_data = extract_idealized_cases(remove_overlapping_samples)\n\nassert len( train_data.loc[train_data['eeg_id'].isin(test_data['eeg_id'].to_list())] ) == 0  # same eeg_id cannot be in train and test samples\nassert len( train_data.loc[train_data['patient_id'].isin(test_data['patient_id'].to_list())] ) == 0 # same patient_id cannot be in train and test samples","metadata":{"execution":{"iopub.status.busy":"2024-03-29T21:51:19.960184Z","iopub.execute_input":"2024-03-29T21:51:19.960557Z","iopub.status.idle":"2024-03-29T21:51:20.377905Z","shell.execute_reply.started":"2024-03-29T21:51:19.960530Z","shell.execute_reply":"2024-03-29T21:51:20.376443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots()\nsns.countplot(train_data, x='expert_consensus', ax=ax)\nax.set_title('Count of expert consensus')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-29T21:51:20.379429Z","iopub.execute_input":"2024-03-29T21:51:20.379800Z","iopub.status.idle":"2024-03-29T21:51:20.661778Z","shell.execute_reply.started":"2024-03-29T21:51:20.379767Z","shell.execute_reply":"2024-03-29T21:51:20.660647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"create_folder('/kaggle/working/idealized/without_overlapping')\n\ntrain_data.to_csv('/kaggle/working/idealized/without_overlapping/train.csv', index=False)\ntest_data.to_csv('/kaggle/working/idealized/without_overlapping/test.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-03-29T21:51:20.663125Z","iopub.execute_input":"2024-03-29T21:51:20.663467Z","iopub.status.idle":"2024-03-29T21:51:20.914911Z","shell.execute_reply.started":"2024-03-29T21:51:20.663438Z","shell.execute_reply":"2024-03-29T21:51:20.913636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Proto Cases","metadata":{}},{"cell_type":"markdown","source":"## With Overlapping Samples","metadata":{}},{"cell_type":"code","source":"remove_overlapping_samples = False\n\ntrain_data, test_data = extract_proto_cases(remove_overlapping_samples)\n\nassert len( train_data.loc[train_data['eeg_id'].isin(test_data['eeg_id'].to_list())] ) == 0  # same eeg_id cannot be in train and test samples\nassert len( train_data.loc[train_data['patient_id'].isin(test_data['patient_id'].to_list())] ) == 0 # same patient_id cannot be in train and test samples","metadata":{"execution":{"iopub.status.busy":"2024-03-29T21:51:20.916491Z","iopub.execute_input":"2024-03-29T21:51:20.916872Z","iopub.status.idle":"2024-03-29T21:51:21.243666Z","shell.execute_reply.started":"2024-03-29T21:51:20.916844Z","shell.execute_reply":"2024-03-29T21:51:21.242767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots()\nsns.countplot(train_data, x='expert_consensus', ax=ax)\nax.set_title('Count of expert consensus')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-29T21:51:21.245322Z","iopub.execute_input":"2024-03-29T21:51:21.246070Z","iopub.status.idle":"2024-03-29T21:51:21.549742Z","shell.execute_reply.started":"2024-03-29T21:51:21.246028Z","shell.execute_reply":"2024-03-29T21:51:21.548788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"create_folder('/kaggle/working/proto/with_overlapping')\n\ntrain_data.to_csv('/kaggle/working/proto/with_overlapping/train.csv', index=False)\ntest_data.to_csv('/kaggle/working/proto/with_overlapping/test.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-03-29T21:51:21.554170Z","iopub.execute_input":"2024-03-29T21:51:21.554991Z","iopub.status.idle":"2024-03-29T21:51:21.637039Z","shell.execute_reply.started":"2024-03-29T21:51:21.554953Z","shell.execute_reply":"2024-03-29T21:51:21.635966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Without Overlapping Samples","metadata":{}},{"cell_type":"code","source":"remove_overlapping_samples = True\n\ntrain_data, test_data = extract_proto_cases(remove_overlapping_samples)\n\nassert len( train_data.loc[train_data['eeg_id'].isin(test_data['eeg_id'].to_list())] ) == 0  # same eeg_id cannot be in train and test samples\nassert len( train_data.loc[train_data['patient_id'].isin(test_data['patient_id'].to_list())] ) == 0 # same patient_id cannot be in train and test samples","metadata":{"execution":{"iopub.status.busy":"2024-03-29T21:51:21.638494Z","iopub.execute_input":"2024-03-29T21:51:21.638864Z","iopub.status.idle":"2024-03-29T21:51:22.005729Z","shell.execute_reply.started":"2024-03-29T21:51:21.638834Z","shell.execute_reply":"2024-03-29T21:51:22.004761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots()\nsns.countplot(train_data, x='expert_consensus', ax=ax)\nax.set_title('Count of expert consensus')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-29T21:51:22.006945Z","iopub.execute_input":"2024-03-29T21:51:22.007265Z","iopub.status.idle":"2024-03-29T21:51:22.297114Z","shell.execute_reply.started":"2024-03-29T21:51:22.007238Z","shell.execute_reply":"2024-03-29T21:51:22.295986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"create_folder('/kaggle/working/proto/without_overlapping')\n\ntrain_data.to_csv('/kaggle/working/proto/without_overlapping/train.csv', index=False)\ntest_data.to_csv('/kaggle/working/proto/without_overlapping/test.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-03-29T21:51:22.298830Z","iopub.execute_input":"2024-03-29T21:51:22.299509Z","iopub.status.idle":"2024-03-29T21:51:22.322776Z","shell.execute_reply.started":"2024-03-29T21:51:22.299471Z","shell.execute_reply":"2024-03-29T21:51:22.321759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Edge Cases","metadata":{}},{"cell_type":"markdown","source":"## With Overlapping Samples","metadata":{}},{"cell_type":"code","source":"remove_overlapping_samples = False\n\ntrain_data, test_data = extract_edge_cases(remove_overlapping_samples)\n\nassert len( train_data.loc[train_data['eeg_id'].isin(test_data['eeg_id'].to_list())] ) == 0  # same eeg_id cannot be in train and test samples\nassert len( train_data.loc[train_data['patient_id'].isin(test_data['patient_id'].to_list())] ) == 0 # same patient_id cannot be in train and test samples","metadata":{"execution":{"iopub.status.busy":"2024-03-29T21:51:22.324217Z","iopub.execute_input":"2024-03-29T21:51:22.324539Z","iopub.status.idle":"2024-03-29T21:51:22.646680Z","shell.execute_reply.started":"2024-03-29T21:51:22.324512Z","shell.execute_reply":"2024-03-29T21:51:22.644888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots()\nsns.countplot(train_data, x='expert_consensus', ax=ax)\nax.set_title('Count of expert consensus')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-29T21:51:22.649046Z","iopub.execute_input":"2024-03-29T21:51:22.650326Z","iopub.status.idle":"2024-03-29T21:51:22.947201Z","shell.execute_reply.started":"2024-03-29T21:51:22.650272Z","shell.execute_reply":"2024-03-29T21:51:22.946061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"create_folder('/kaggle/working/edge/with_overlapping')\n\ntrain_data.to_csv('/kaggle/working/edge/with_overlapping/train.csv', index=False)\ntest_data.to_csv('/kaggle/working/edge/with_overlapping/test.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-03-29T21:51:22.949134Z","iopub.execute_input":"2024-03-29T21:51:22.949979Z","iopub.status.idle":"2024-03-29T21:51:22.991009Z","shell.execute_reply.started":"2024-03-29T21:51:22.949938Z","shell.execute_reply":"2024-03-29T21:51:22.989765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Without Overlapping Samples","metadata":{}},{"cell_type":"code","source":"remove_overlapping_samples = True\n\ntrain_data, test_data = extract_edge_cases(remove_overlapping_samples)\n\nassert len( train_data.loc[train_data['eeg_id'].isin(test_data['eeg_id'].to_list())] ) == 0  # same eeg_id cannot be in train and test samples\nassert len( train_data.loc[train_data['patient_id'].isin(test_data['patient_id'].to_list())] ) == 0 # same patient_id cannot be in train and test samples","metadata":{"execution":{"iopub.status.busy":"2024-03-29T21:51:22.992304Z","iopub.execute_input":"2024-03-29T21:51:22.992632Z","iopub.status.idle":"2024-03-29T21:51:23.364402Z","shell.execute_reply.started":"2024-03-29T21:51:22.992605Z","shell.execute_reply":"2024-03-29T21:51:23.363435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots()\nsns.countplot(train_data, x='expert_consensus', ax=ax)\nax.set_title('Count of expert consensus')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-29T21:51:23.365721Z","iopub.execute_input":"2024-03-29T21:51:23.366068Z","iopub.status.idle":"2024-03-29T21:51:23.573788Z","shell.execute_reply.started":"2024-03-29T21:51:23.366040Z","shell.execute_reply":"2024-03-29T21:51:23.572560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"create_folder('/kaggle/working/edge/without_overlapping')\n\ntrain_data.to_csv('/kaggle/working/edge/without_overlapping/train.csv', index=False)\ntest_data.to_csv('/kaggle/working/edge/without_overlapping/test.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-03-29T21:51:23.575090Z","iopub.execute_input":"2024-03-29T21:51:23.575413Z","iopub.status.idle":"2024-03-29T21:51:23.589012Z","shell.execute_reply.started":"2024-03-29T21:51:23.575385Z","shell.execute_reply":"2024-03-29T21:51:23.587523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}