{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":29653,"databundleVersionId":2420395,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install openpyxl pydicom","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T10:21:45.664961Z","iopub.execute_input":"2025-07-17T10:21:45.665246Z","iopub.status.idle":"2025-07-17T10:21:50.937760Z","shell.execute_reply.started":"2025-07-17T10:21:45.665225Z","shell.execute_reply":"2025-07-17T10:21:50.936420Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"运行2脱敏","metadata":{}},{"cell_type":"code","source":"import os\nimport pydicom\nfrom pydicom.dataset import Dataset\n\ndef anonymize_dicom(input_file, output_file=None, overwrite=False):\n    \n    try:\n        # 读取DICOM文件\n        ds = pydicom.dcmread(input_file)\n        \n        # 定义需要脱敏的DICOM标签映射\n        tags_to_anonymize = {\n            # 患者信息\n            \"PatientName\": \"Anonymous\",\n            \"PatientID\": \"Anonymous\",\n            \"PatientBirthDate\": \"Anonymous\",\n            \"PatientBirthTime\": \"Anonymous\",\n            \"OtherPatientIDs\": \"Anonymous\",\n            \"OtherPatientNames\": \"Anonymous\",\n            \"PatientBirthName\": \"Anonymous\",\n            \"PatientSize\": \"Anonymous\",\n            \"PatientWeight\": \"Anonymous\",\n            \"PatientAddress\": \"Anonymous\",\n            \"PatientTelephoneNumbers\": \"Anonymous\",\n            \n            # 医疗机构和医生信息\n            \"InstitutionName\": \"Anonymous\",\n            \"InstitutionAddress\": \"Anonymous\",\n            \"ReferringPhysicianName\": \"Anonymous\",\n            \"ReferringPhysicianAddress\": \"Anonymous\",\n            \"ReferringPhysicianTelephoneNumbers\": \"Anonymous\",\n            \"InstitutionalDepartmentName\": \"Anonymous\",\n            \"PerformingPhysicianName\": \"Anonymous\",\n            \"OperatorsName\": \"Anonymous\",\n        }\n        \n        # 对每个标签进行脱敏处理\n        for tag, value in tags_to_anonymize.items():\n            if tag in ds:\n                setattr(ds, tag, value)\n                #print(f\"已脱敏: {tag}\")\n        \n        # 特殊处理：Other Patient IDs Sequence (0010,1002)\n        if \"OtherPatientIDsSequence\" in ds:\n            # 创建一个新的空序列\n            new_sequence = pydicom.sequence.Sequence()\n            \n            # 创建一个新的数据集\n            item = Dataset()\n            item.PatientID = \"Anonymous\"  # 设置序列中可能包含的患者ID字段\n            \n            # 将数据集添加到序列中\n            new_sequence.append(item)\n            \n            # 替换原有序列\n            ds.OtherPatientIDsSequence = new_sequence\n            #print(\"已脱敏: OtherPatientIDsSequence\")\n        \n        # 保存处理后的文件\n        if output_file:\n            ds.save_as(output_file)\n        elif overwrite and False:\n            ds.save_as(input_file)\n        else:\n            # 如果没有提供输出路径且不允许覆盖，则在原文件名后添加\"_anon\"\n            base, ext = os.path.splitext(input_file)\n            output_file = f\"{base}_anon{ext}\"\n            ds.save_as(output_file)\n        \n        #print(f\"脱敏完成，文件已保存至: {output_file if output_file else input_file}\")\n        return True\n    except Exception as e:\n        print(f\"处理文件时出错: {input_file}\")\n        print(f\"错误信息: {str(e)}\")\n        return False\n\n\n\noutputpath0 = \"/kaggle/working/rsna-miccai-brain-tumor-radiogenomic-classification/trainanonymized\"\nfilepath0 = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\nfor n,filepath1 in enumerate(os.listdir(filepath0)):\n    if n>=2:break\n    for filepath2 in os.listdir(os.path.join(filepath0,filepath1)):\n        for filepath3 in os.listdir(os.path.join(filepath0,filepath1,filepath2)):\n            output_path = os.path.join(outputpath0,filepath1,filepath2,filepath3)\n            input_path = os.path.join(filepath0,filepath1,filepath2,filepath3)\n            mkdir = os.path.join(outputpath0,filepath1,filepath2)\n            os.makedirs(mkdir,exist_ok=True)\n            r = anonymize_dicom(input_path, output_path, False)    \n            print(mkdir,r)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-18T03:58:17.453056Z","iopub.execute_input":"2025-07-18T03:58:17.453462Z","iopub.status.idle":"2025-07-18T03:58:24.920634Z","shell.execute_reply.started":"2025-07-18T03:58:17.453430Z","shell.execute_reply":"2025-07-18T03:58:24.919810Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"测试脱敏","metadata":{}},{"cell_type":"code","source":"testcase = '/kaggle/working/rsna-miccai-brain-tumor-radiogenomic-classification/trainanonymized/00058/FLAIR/Image-168.dcm'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-18T03:59:48.134006Z","iopub.execute_input":"2025-07-18T03:59:48.134897Z","iopub.status.idle":"2025-07-18T03:59:48.140418Z","shell.execute_reply.started":"2025-07-18T03:59:48.134840Z","shell.execute_reply":"2025-07-18T03:59:48.139279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ds = pydicom.dcmread(testcase)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-18T04:00:22.262735Z","iopub.execute_input":"2025-07-18T04:00:22.263769Z","iopub.status.idle":"2025-07-18T04:00:22.272928Z","shell.execute_reply.started":"2025-07-18T04:00:22.263717Z","shell.execute_reply":"2025-07-18T04:00:22.271689Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-18T04:00:26.761478Z","iopub.execute_input":"2025-07-18T04:00:26.761879Z","iopub.status.idle":"2025-07-18T04:00:26.774097Z","shell.execute_reply.started":"2025-07-18T04:00:26.761846Z","shell.execute_reply":"2025-07-18T04:00:26.773061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pydicom\nimport shutil\nimport pandas as pd\nfrom pydicom.dataset import Dataset\nfrom typing import List, Dict, Set, Tuple, Optional, Any\nfrom openpyxl.styles import PatternFill, Font\n\n\nclass DicomDataCleaner:\n    def __init__(self, input_dir: str, output_dir: str):\n        \"\"\"初始化DICOM数据清洗器\"\"\"\n        self.input_dir = input_dir\n        self.output_dir = output_dir\n        self.allowed_magnetic_fields = {\"1.5\", \"3.0\"}  # 允许的磁场强度(T)\n        self.max_slice_thickness = 5.0  # 最大允许层厚(mm)\n        self.min_b_value = 1400  # 最小B值\n        self.processed_study_uids = set()  # 存储已处理的Study UID\n        self.cleaning_details = {}  # 存储详细清理结果（按条件记录）\n        \n        # 创建输出目录\n        os.makedirs(output_dir, exist_ok=True)\n    \n    # 以下为原有工具方法（保持不变）\n    def _is_dicom_readable(self, file_path: str) -> bool:\n        try:\n            pydicom.dcmread(file_path)\n            return True\n        except Exception:\n            return False\n    \n    def _is_mr_data(self, ds: Dataset) -> bool:\n        return ds.Modality.lower() == \"mr\" if \"Modality\" in ds else False\n    \n    def _check_unique_study_uid(self, study_uid: str) -> bool:\n        return study_uid not in self.processed_study_uids\n    \n    def _has_required_tags(self, ds: Dataset) -> bool:\n        required_tags = [\n            \"StudyInstanceUID\", \"SeriesInstanceUID\", \"SOPInstanceUID\",\n            \"Modality\", \"PatientID\", \"StudyDate\", \"SeriesDescription\"\n        ]\n        for tag in required_tags:\n            if tag not in ds:\n                return False\n        return True\n    \n    def _classify_series(self, ds: Dataset) -> Optional[str]:\n        series_desc = ds.get(\"SeriesDescription\", \"\").lower()\n        \n        if \"dwi\" in series_desc or \"diffusion\" in series_desc:\n            if \"adc\" in series_desc:\n                return \"adc\"\n            return \"dwi\"\n        \n        if \"t2\" in series_desc and self._is_traverse_plane(ds):\n            tr = self._get_float_value(ds, \"RepetitionTime\")\n            te = self._get_float_value(ds, \"EchoTime\")\n            if tr is not None and te is not None and tr > 2000 and te > 75:\n                return \"t2_tra\"\n        return None\n    \n    def _get_float_value(self, ds: Dataset, tag_name: str) -> Optional[float]:\n        value = ds.get(tag_name)\n        if value is None:\n            return None\n        try:\n            return float(value)\n        except (ValueError, TypeError):\n            return None\n    \n    def _is_traverse_plane(self, ds: Dataset) -> bool:\n        if \"ImageOrientationPatient\" in ds:\n            orientation = ds.ImageOrientationPatient\n            zx, zy, zz = orientation[3], orientation[4], orientation[5]\n            return abs(zz) > 0.9\n        return False\n    \n    def _get_b_value(self, ds: Dataset) -> Optional[int]:\n        if \"DiffusionBValue\" in ds:\n            try:\n                return int(ds.DiffusionBValue)\n            except (ValueError, TypeError):\n                pass\n        series_desc = ds.get(\"SeriesDescription\", \"\").lower()\n        for part in series_desc.split():\n            if \"b\" in part and part.replace(\"b\", \"\").isdigit():\n                return int(part.replace(\"b\", \"\"))\n        return None\n    \n    def _check_series_integrity(self, series_files: List[str]) -> bool:\n        sop_uids = set()\n        instance_numbers = set()\n        for file in series_files:\n            try:\n                ds = pydicom.dcmread(file)\n                if ds.SOPInstanceUID in sop_uids:\n                    return False\n                sop_uids.add(ds.SOPInstanceUID)\n                if \"InstanceNumber\" in ds:\n                    instance_numbers.add(ds.InstanceNumber)\n            except Exception:\n                return False\n        if instance_numbers:\n            min_num = min(instance_numbers)\n            max_num = max(instance_numbers)\n            if len(instance_numbers) != (max_num - min_num + 1):\n                return False\n        return True\n    \n    def _get_magnetic_field_strength(self, ds: Dataset) -> Optional[str]:\n        if \"MagneticFieldStrength\" in ds:\n            return str(ds.MagneticFieldStrength)\n        return None\n    \n    def _get_slice_thickness(self, ds: Dataset) -> Optional[float]:\n        return self._get_float_value(ds, \"SliceThickness\")\n    \n    def process_study(self, study_dir: str) -> Tuple[bool, Dict[str, Any]]:\n        \"\"\"处理单个Study，返回(是否有效, 详细条件检查结果)\"\"\"\n        study_name = os.path.basename(study_dir)\n        # 初始化详细检查结果（所有条件默认False）\n        check_results = {\n            \"有可读DICOM文件\": False,\n            \"是MR数据\": False,\n            \"StudyUID唯一\": False,\n            \"包含T2横断面序列\": False,\n            \"包含ADC序列\": False,\n            \"包含DWI序列\": False,\n            \"T2序列完整\": False,\n            \"ADC序列完整\": False,\n            \"DWI序列完整\": False,\n            \"DWI包含有效B值\": False,\n            \"T2磁场强度合规\": False,\n            \"ADC磁场强度合规\": False,\n            \"DWI磁场强度合规\": False,\n            \"T2层厚合规\": False,\n            \"ADC层厚合规\": False,\n            \"DWI层厚合规\": False,\n            \"总体有效\": False,\n            \"失败原因\": []\n        }\n        \n        # 1. 检查是否有可读DICOM文件\n        dicom_files = []\n        for root, _, files in os.walk(study_dir):\n            for file in files:\n                file_path = os.path.join(root, file)\n                if self._is_dicom_readable(file_path):\n                    dicom_files.append(file_path)\n        if not dicom_files:\n            check_results[\"失败原因\"].append(\"Study目录中没有可读的DICOM文件\")\n            self.cleaning_details[study_name] = check_results\n            return False, check_results\n        check_results[\"有可读DICOM文件\"] = True\n        \n        # 2. 检查是否为MR数据\n        first_ds = pydicom.dcmread(dicom_files[0])\n        if not self._is_mr_data(first_ds):\n            check_results[\"失败原因\"].append(\"非MR数据\")\n            self.cleaning_details[study_name] = check_results\n            return False, check_results\n        check_results[\"是MR数据\"] = True\n        \n        # 3. 检查StudyUID唯一性\n        study_uid = first_ds.StudyInstanceUID\n        if not self._check_unique_study_uid(study_uid):\n            check_results[\"失败原因\"].append(f\"重复的StudyInstanceUID: {study_uid}\")\n            self.cleaning_details[study_name] = check_results\n            return False, check_results\n        check_results[\"StudyUID唯一\"] = True\n        self.processed_study_uids.add(study_uid)\n        \n        # 4. 按Series分组并分类\n        series_groups = {}\n        for file in dicom_files:\n            ds = pydicom.dcmread(file)\n            series_uid = ds.SeriesInstanceUID\n            if series_uid not in series_groups:\n                series_groups[series_uid] = []\n            series_groups[series_uid].append(file)\n        \n        t2_tra_series = []\n        adc_series = []\n        dwi_series = []\n        for series_uid, files in series_groups.items():\n            ds = pydicom.dcmread(files[0])\n            series_type = self._classify_series(ds)\n            if series_type == \"t2_tra\":\n                t2_tra_series.append((series_uid, files))\n            elif series_type == \"adc\":\n                adc_series.append((series_uid, files))\n            elif series_type == \"dwi\":\n                dwi_series.append((series_uid, files))\n        \n        # 5. 检查是否包含必要序列\n        check_results[\"包含T2横断面序列\"] = len(t2_tra_series) > 0\n        check_results[\"包含ADC序列\"] = len(adc_series) > 0\n        check_results[\"包含DWI序列\"] = len(dwi_series) > 0\n        if not all([check_results[\"包含T2横断面序列\"], \n                   check_results[\"包含ADC序列\"], \n                   check_results[\"包含DWI序列\"]]):\n            check_results[\"失败原因\"].append(\"缺少必要序列（T2/ADC/DWI）\")\n            self.cleaning_details[study_name] = check_results\n            return False, check_results\n        \n        # 6. 检查序列完整性\n        valid_t2 = [f for s, f in t2_tra_series if self._check_series_integrity(f)]\n        valid_adc = [f for s, f in adc_series if self._check_series_integrity(f)]\n        valid_dwi = [f for s, f in dwi_series if self._check_series_integrity(f)]\n        \n        check_results[\"T2序列完整\"] = len(valid_t2) > 0\n        check_results[\"ADC序列完整\"] = len(valid_adc) > 0\n        check_results[\"DWI序列完整\"] = len(valid_dwi) > 0\n        if not all([check_results[\"T2序列完整\"], \n                   check_results[\"ADC序列完整\"], \n                   check_results[\"DWI序列完整\"]]):\n            check_results[\"失败原因\"].append(\"存在不完整的序列\")\n            self.cleaning_details[study_name] = check_results\n            return False, check_results\n        \n        # 7. 检查DWI的B值\n        valid_dwi_series = []\n        for series_uid, files in valid_dwi:\n            has_valid_b = any(self._get_b_value(pydicom.dcmread(f)) >= self.min_b_value \n                             for f in files if self._get_b_value(pydicom.dcmread(f)) is not None)\n            if has_valid_b:\n                valid_dwi_series.append((series_uid, files))\n        check_results[\"DWI包含有效B值\"] = len(valid_dwi_series) > 0\n        if not check_results[\"DWI包含有效B值\"]:\n            check_results[\"失败原因\"].append(f\"DWI缺少b值>={self.min_b_value}的图像\")\n            self.cleaning_details[study_name] = check_results\n            return False, check_results\n        \n        # 8. 检查磁场强度\n        t2_field_ok = all(self._get_magnetic_field_strength(pydicom.dcmread(f[0])) in self.allowed_magnetic_fields \n                         for f in valid_t2)\n        adc_field_ok = all(self._get_magnetic_field_strength(pydicom.dcmread(f[0])) in self.allowed_magnetic_fields \n                         for f in valid_adc)\n        dwi_field_ok = all(self._get_magnetic_field_strength(pydicom.dcmread(f[0])) in self.allowed_magnetic_fields \n                         for f in valid_dwi_series)\n        \n        check_results[\"T2磁场强度合规\"] = t2_field_ok\n        check_results[\"ADC磁场强度合规\"] = adc_field_ok\n        check_results[\"DWI磁场强度合规\"] = dwi_field_ok\n        if not all([t2_field_ok, adc_field_ok, dwi_field_ok]):\n            check_results[\"失败原因\"].append(\"磁场强度不符合要求（需1.5T或3.0T）\")\n            self.cleaning_details[study_name] = check_results\n            return False, check_results\n        \n        # 9. 检查层厚\n        t2_thickness_ok = all(self._get_slice_thickness(pydicom.dcmread(f[0])) is not None \n                            and self._get_slice_thickness(pydicom.dcmread(f[0])) <= self.max_slice_thickness \n                            for f in valid_t2)\n        adc_thickness_ok = all(self._get_slice_thickness(pydicom.dcmread(f[0])) is not None \n                             and self._get_slice_thickness(pydicom.dcmread(f[0])) <= self.max_slice_thickness \n                             for f in valid_adc)\n        dwi_thickness_ok = all(self._get_slice_thickness(pydicom.dcmread(f[0])) is not None \n                             and self._get_slice_thickness(pydicom.dcmread(f[0])) <= self.max_slice_thickness \n                             for f in valid_dwi_series)\n        \n        check_results[\"T2层厚合规\"] = t2_thickness_ok\n        check_results[\"ADC层厚合规\"] = adc_thickness_ok\n        check_results[\"DWI层厚合规\"] = dwi_thickness_ok\n        if not all([t2_thickness_ok, adc_thickness_ok, dwi_thickness_ok]):\n            check_results[\"失败原因\"].append(f\"层厚超过限制（最大{self.max_slice_thickness}mm）\")\n            self.cleaning_details[study_name] = check_results\n            return False, check_results\n        \n        # 所有检查通过\n        check_results[\"总体有效\"] = True\n        check_results[\"失败原因\"] = [\"无\"]\n        self._save_valid_series(study_uid, valid_t2, valid_adc, valid_dwi_series)\n        self.cleaning_details[study_name] = check_results\n        return True, check_results\n    \n    def _save_valid_series(self, study_uid: str, t2_series: List, adc_series: List, dwi_series: List) -> None:\n        study_output_dir = os.path.join(self.output_dir, study_uid)\n        os.makedirs(study_output_dir, exist_ok=True)\n        \n        for i, (series_uid, files) in enumerate(t2_series):\n            series_dir = os.path.join(study_output_dir, f\"T2_TRA_{i+1}\")\n            os.makedirs(series_dir, exist_ok=True)\n            for file in files:\n                shutil.copy(file, os.path.join(series_dir, os.path.basename(file)))\n        \n        for i, (series_uid, files) in enumerate(adc_series):\n            series_dir = os.path.join(study_output_dir, f\"ADC_{i+1}\")\n            os.makedirs(series_dir, exist_ok=True)\n            for file in files:\n                shutil.copy(file, os.path.join(series_dir, os.path.basename(file)))\n        \n        for i, (series_uid, files) in enumerate(dwi_series):\n            series_dir = os.path.join(study_output_dir, f\"DWI_b{self.min_b_value}_{i+1}\")\n            os.makedirs(series_dir, exist_ok=True)\n            for file in files:\n                shutil.copy(file, os.path.join(series_dir, os.path.basename(file)))\n    \n    def process_directory(self) -> pd.DataFrame:\n        print(f\"开始处理目录: {self.input_dir}\")\n        study_dirs = [os.path.join(self.input_dir, d) for d in os.listdir(self.input_dir) \n                     if os.path.isdir(os.path.join(self.input_dir, d))]\n        \n        total_studies = len(study_dirs)\n        valid_studies = 0\n        \n        for i, study_dir in enumerate(study_dirs, 1):\n            study_name = os.path.basename(study_dir)\n            print(f\"\\n处理Study {i}/{total_studies}: {study_name}\")\n            is_valid, _ = self.process_study(study_dir)\n            if is_valid:\n                valid_studies += 1\n        \n        print(f\"\\n处理完成!\")\n        print(f\"总Studies: {total_studies}\")\n        print(f\"有效Studies: {valid_studies}\")\n        print(f\"无效Studies: {total_studies - valid_studies}\")\n        print(f\"有效数据已保存至: {self.output_dir}\")\n        \n        return self.get_report_dataframe()\n    \n    def get_report_dataframe(self) -> pd.DataFrame:\n        \"\"\"生成包含所有条件列的DataFrame\"\"\"\n        data = []\n        for study_name, check_results in self.cleaning_details.items():\n            # 拼接失败原因（多个原因用换行分隔）\n            failure_reasons = \"\\n\".join(check_results[\"失败原因\"]) if check_results[\"失败原因\"] else \"\"\n            # 构造一行数据（Study名称+所有检查条件）\n            row_data = {\"Study名称\": study_name}\n            row_data.update(check_results)  # 合并所有条件检查结果\n            row_data[\"失败原因\"] = failure_reasons  # 替换列表为字符串\n            data.append(row_data)\n        \n        return pd.DataFrame(data)\n    \n    def save_report_to_excel(self, excel_path: str) -> None:\n        report_df = self.get_report_dataframe()\n        \n        with pd.ExcelWriter(excel_path, engine='openpyxl') as writer:\n            report_df.to_excel(writer, sheet_name='详细检查报告', index=False)\n            worksheet = writer.sheets['详细检查报告']\n            \n            # 调整列宽\n            for i, col in enumerate(report_df.columns):\n                max_length = max(report_df[col].astype(str).map(len).max(), len(col)) + 2\n                worksheet.column_dimensions[chr(65 + i)].width = max_length\n            \n            # 定义样式（True为绿色，False为红色）","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-18T03:25:03.221662Z","iopub.execute_input":"2025-07-18T03:25:03.222033Z","iopub.status.idle":"2025-07-18T03:25:04.825015Z","shell.execute_reply.started":"2025-07-18T03:25:03.222009Z","shell.execute_reply":"2025-07-18T03:25:04.824086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ninputpath0 = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\noutputpath0 = \"/kaggle/working/rsna-miccai-brain-tumor-radiogenomic-classification/train_cleaned\"\n\nfor n,i in enumerate(os.listdir(inputpath0)):\n    if n>=20:break\n    inputpath1 = os.path.join(inputpath0,i)\n    outputpath1 = os.path.join(outputpath0,i)\n    excel = outputpath0+\"//\"+i+\".xlsx\"\n    cleaner = DicomDataCleaner(inputpath1, outputpath1)\n    report_df = cleaner.process_directory()\n    cleaner.save_report_to_excel(excel)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-18T03:25:05.771352Z","iopub.execute_input":"2025-07-18T03:25:05.772096Z","iopub.status.idle":"2025-07-18T03:27:52.785927Z","shell.execute_reply.started":"2025-07-18T03:25:05.772067Z","shell.execute_reply":"2025-07-18T03:27:52.785060Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\ncleanreportcase = \"\"\"/kaggle/working/rsna-miccai-brain-tumor-radiogenomic-classification/train_cleaned/00688.xlsx\"\"\"\nexcelfile = pd.ExcelFile(cleanreportcase)\nexcelfile.parse(excelfile.sheet_names[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-18T04:02:04.324529Z","iopub.execute_input":"2025-07-18T04:02:04.324971Z","iopub.status.idle":"2025-07-18T04:02:04.506275Z","shell.execute_reply.started":"2025-07-18T04:02:04.324940Z","shell.execute_reply":"2025-07-18T04:02:04.505376Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}