{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":10943904,"sourceType":"datasetVersion","datasetId":6806501}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport random\nfrom sklearn.model_selection import GroupShuffleSplit","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:21:45.714201Z","iopub.execute_input":"2025-03-06T16:21:45.714591Z","iopub.status.idle":"2025-03-06T16:21:46.871543Z","shell.execute_reply.started":"2025-03-06T16:21:45.714560Z","shell.execute_reply":"2025-03-06T16:21:46.870148Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PROJECT_DIR = '/kaggle/input'\nOUTPUT_DIR = '/kaggle/working'\nMETADATA_DIR = os.path.join('/kaggle/input/rsna-lumbar-metadata', 'data', 'processed_metadata')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:31:27.496876Z","iopub.execute_input":"2025-03-06T16:31:27.497449Z","iopub.status.idle":"2025-03-06T16:31:27.502831Z","shell.execute_reply.started":"2025-03-06T16:31:27.497406Z","shell.execute_reply":"2025-03-06T16:31:27.501374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Ensure deterministic behavior\nrandom.seed(hash(\"setting random seeds\") % 2**32 - 1)\nnp.random.seed(hash(\"improves reproducibility\") % 2**32 - 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:22:07.632788Z","iopub.execute_input":"2025-03-06T16:22:07.633196Z","iopub.status.idle":"2025-03-06T16:22:07.638321Z","shell.execute_reply.started":"2025-03-06T16:22:07.633152Z","shell.execute_reply":"2025-03-06T16:22:07.637041Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(os.path.join(METADATA_DIR, 'processed_metadata.csv'))\ncondition_types = df['condition'].unique().tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:28:13.944086Z","iopub.execute_input":"2025-03-06T16:28:13.944611Z","iopub.status.idle":"2025-03-06T16:28:14.203139Z","shell.execute_reply.started":"2025-03-06T16:28:13.944566Z","shell.execute_reply":"2025-03-06T16:28:14.201738Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def group_split(df, group_col, test_size=0.15, random_state=42):\n    splitter = GroupShuffleSplit(test_size=test_size, n_splits=1, random_state=random_state)\n    split = splitter.split(df, groups=df[group_col])\n    train_idx, test_idx = next(split)\n    return df.iloc[train_idx], df.iloc[test_idx]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:28:15.908345Z","iopub.execute_input":"2025-03-06T16:28:15.908764Z","iopub.status.idle":"2025-03-06T16:28:15.914682Z","shell.execute_reply.started":"2025-03-06T16:28:15.908728Z","shell.execute_reply":"2025-03-06T16:28:15.913401Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for condition in condition_types:\n    print(condition, \":\")\n    # Remove spaces to write into file names\n    condition = condition.replace(\" \", \"\")\n    \n    # Read in the dataset exclusive to the condition\n    df_condition = pd.read_csv(os.path.join(METADATA_DIR, 'processed_metadata_' + condition + '.csv'))\n    \n    # Split into train, validation and test\n    train_val_df, test_df = group_split(df_condition, 'study_id')\n    train_df, val_df = group_split(train_val_df, 'study_id')\n    print(f\"Number of samples in training set: {train_df.shape[0]}\")\n    print(f\"Number of samples in validation set: {val_df.shape[0]}\")\n    print(f\"Number of samples in test set: {test_df.shape[0]}\")\n    \n    # Write out the splitted subsets\n    WRITE_DIR = os.path.join(OUTPUT_DIR, condition)\n    os.makedirs(WRITE_DIR, exist_ok=True)\n\n    train_df.to_csv(os.path.join(WRITE_DIR, 'train.csv'), index=False)\n    val_df.to_csv  (os.path.join(WRITE_DIR, 'val.csv'),   index=False)\n    test_df.to_csv (os.path.join(WRITE_DIR, 'test.csv'),  index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:31:30.256908Z","iopub.execute_input":"2025-03-06T16:31:30.257405Z","iopub.status.idle":"2025-03-06T16:31:30.997805Z","shell.execute_reply.started":"2025-03-06T16:31:30.257361Z","shell.execute_reply":"2025-03-06T16:31:30.996675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!zip -r RSNA_Lumbar_01_test_train_split.zip /kaggle/working","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:32:26.651591Z","iopub.execute_input":"2025-03-06T16:32:26.651963Z","iopub.status.idle":"2025-03-06T16:32:27.062471Z","shell.execute_reply.started":"2025-03-06T16:32:26.651934Z","shell.execute_reply":"2025-03-06T16:32:27.060860Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:32:31.586196Z","iopub.execute_input":"2025-03-06T16:32:31.586622Z","iopub.status.idle":"2025-03-06T16:32:31.717805Z","shell.execute_reply.started":"2025-03-06T16:32:31.586581Z","shell.execute_reply":"2025-03-06T16:32:31.716443Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}