{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nimport pandas as pd\n\nfrom typing import Any\nfrom dataclasses import dataclass\n","metadata":{"execution":{"iopub.status.busy":"2024-08-26T09:25:47.937369Z","iopub.execute_input":"2024-08-26T09:25:47.937802Z","iopub.status.idle":"2024-08-26T09:25:48.408494Z","shell.execute_reply.started":"2024-08-26T09:25:47.937760Z","shell.execute_reply":"2024-08-26T09:25:48.407452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n@dataclass(slots = True)\nclass DataConfig:\n    ## directories\n    data_root: str = \"./data_store\"\n    dump_root: str = \"./dump/data\"\n    ## fixed transformations\n    voi_lut: bool = True\n    fix_monochrome: bool = False\n    apply_volume_norm: bool = True\n    fixed_transformations: list[str] = None\n    fixed_transformations_kwargs: dict[str, dict[str, Any]] = None\n    fixed_target_transformations: list[str] = None\n    fixed_target_transformations_kwargs: dict[str, dict[str, Any]] = None\n    ## in-memory transformations\n    train_data_transformations: list[str] = None\n    train_data_transformations_kwargs: dict[str, dict[str, Any]] = None\n    val_data_transformations: list[str] = None\n    val_data_transformations_kwargs: dict[str, dict[str, Any]] = None\n    test_data_transformations: list[str] = None\n    test_data_transformations_kwargs: dict[str, dict[str, Any]] = None\n    scalar_target_transformations: list[str] = None\n    scalar_target_transformations_kwargs: dict[str, dict[str, Any]] = None\n    categorical_target_transformations: list[str] = None\n    categorical_target_transformations_kwargs: dict[str, dict[str, Any]] = None\n    ## batch size\n    train_batch_size: int = 48\n    val_batch_size: int = 24\n    test_batch_size: int = 24\n    ## data loader options\n    num_workers: int = 19\n\n    @property\n    def train_images_dir(self):\n        return os.path.join(self.data_root, \"train_images\")\n    \n    @property\n    def test_images_dir(self):\n        return os.path.join(self.data_root, \"test_images\")\n\n    @property\n    def series_description_csv(self):\n        return os.path.join(self.data_root, \"train_series_descriptions.csv\")\n            \n    @property\n    def label_coordinates_csv(self):\n        return os.path.join(self.data_root, \"train_label_coordinates.csv\")\n    \n    @property\n    def labels_csv(self):\n        return os.path.join(self.data_root, \"train.csv\")\n\n    \nclass Metadata:\n    def __init__(\n            self,\n            config\n        ):\n        self.config = config\n        self.series_descriptions_df, self.label_coordinates_df, self.labels_df = self.load_data_frames()\n        ## defining the map for the catgorical variables\n        self.labels_map = {'Normal/Mild': 0, 'Moderate': 1, 'Severe': 2} \n        self.conditions_map = {condition: idx for idx, condition in enumerate(self.labels_df.columns[1:])}\n        ## encoding labels\n        self.encode_labels()\n        ## filling nan values\n        self.fill_na()\n        ## remove missing data\n        self.remove_missing_series_ids()\n        ## merging the data frame\n        self.merged_df = self.get_merged_df()\n        ## aggregating condition and level fields\n        self.aggregate_condition_and_level()\n        self.encode_conditions()\n\n    def load_data_frames(\n            self,\n        ):\n        series_descriptions_df = pd.read_csv(self.config.series_description_csv)\n        label_coordinates_df = pd.read_csv(self.config.label_coordinates_csv)\n        labels_df = pd.read_csv(self.config.labels_csv)\n        return series_descriptions_df, label_coordinates_df, labels_df\n    \n    def fill_na(\n            self,\n        ):\n        self.series_descriptions_df.fillna(0)\n        self.labels_df.fillna(0)\n        self.label_coordinates_df.fillna(0)\n    \n    def encode_labels(\n            self,\n        ):\n        encode_label = lambda label: self.labels_map[label] if isinstance(label, str) else label\n        self.labels_df = self.labels_df.map(encode_label)\n\n    def encode_conditions(\n            self,\n        ):\n        ## encoding conditions \n        encode_label = lambda label: self.conditions_map[label] if isinstance(label, str) else label\n        self.merged_df[\"encoded_condition\"] = self.merged_df[\"full_condition\"].map(encode_label)\n    \n    def get_missing_series_ids(\n            self\n        ):\n        missing_series_ids = set(self.series_descriptions_df[\"series_id\"].unique()).difference(self.label_coordinates_df[\"series_id\"].unique())\n        return missing_series_ids\n\n    def get_missing_study_ids(\n            self\n        ):\n        missing_study_ids = set(self.series_descriptions_df[\"study_id\"].unique()).difference(self.label_coordinates_df[\"study_id\"].unique())\n        return missing_study_ids\n    \n    def remove_missing_series_ids(\n            self\n        ):\n        missing_series_ids = self.get_missing_series_ids()\n        for missing_series_id in missing_series_ids:\n            self.series_descriptions_df = self.series_descriptions_df[self.series_descriptions_df.series_id != missing_series_id] \n\n    def remove_missing_study_ids(\n            self\n        ):\n        missing_study_ids = self.get_missing_study_ids()\n        for missing_study_id in missing_study_ids:\n            self.series_descriptions_df = self.series_descriptions_df[self.series_descriptions_df.study_id != missing_study_id] \n        \n    def get_merged_df(\n            self\n        ):\n        merged_df = pd.merge(self.series_descriptions_df, self.label_coordinates_df, on = \"series_id\")\n        merged_df = merged_df.rename({\"study_id_x\": \"study_id\"}, axis = \"columns\")\n        merged_df = merged_df.drop(\"study_id_y\", axis = \"columns\")\n        return merged_df\n    \n    def aggregate_condition_and_level(\n            self\n        ):\n        self.merged_df[\"condition\"] = self.merged_df[\"condition\"].map(lambda s: s.lower())\n        self.merged_df[\"condition\"] = self.merged_df[\"condition\"].map(lambda s: s.replace(\" \", \"_\"))\n        self.merged_df[\"level\"] = self.merged_df[\"level\"].map(lambda s: s.lower())\n        self.merged_df[\"level\"] = self.merged_df[\"level\"].map(lambda s: s.replace(\"/\", \"_\"))\n        self.merged_df[\"full_condition\"] = self.merged_df[\"condition\"] + \"_\" + self.merged_df[\"level\"] \n","metadata":{"execution":{"iopub.status.busy":"2024-08-26T09:26:14.649556Z","iopub.execute_input":"2024-08-26T09:26:14.650848Z","iopub.status.idle":"2024-08-26T09:26:14.685693Z","shell.execute_reply.started":"2024-08-26T09:26:14.650790Z","shell.execute_reply":"2024-08-26T09:26:14.684533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## retrieving the studies\ndata_root = \"/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification\"\ntest_data = os.path.join(data_root, \"test_images\")\ntest_studies = os.listdir(test_data)","metadata":{"execution":{"iopub.status.busy":"2024-08-26T09:26:15.737815Z","iopub.execute_input":"2024-08-26T09:26:15.738623Z","iopub.status.idle":"2024-08-26T09:26:15.747845Z","shell.execute_reply.started":"2024-08-26T09:26:15.738575Z","shell.execute_reply":"2024-08-26T09:26:15.746663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## initializing metadata\nmetadata_config = DataConfig(data_root = data_root, dump_root = \"/kaggle/working/\")\nmetadata = Metadata(metadata_config)\nconditions = sorted(metadata.conditions_map.keys())\nlabels = {\"normal_mild\", \"moderate\", \"severe\"}","metadata":{"execution":{"iopub.status.busy":"2024-08-26T09:26:33.433435Z","iopub.execute_input":"2024-08-26T09:26:33.434566Z","iopub.status.idle":"2024-08-26T09:26:33.884654Z","shell.execute_reply.started":"2024-08-26T09:26:33.434517Z","shell.execute_reply":"2024-08-26T09:26:33.883517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## inferring frequencies\nvalue_counts = dict()\ncolumns = metadata.labels_df.columns[1:]\nfor column in columns:\n    value_counts[column] = dict(metadata.labels_df[column].value_counts(normalize=True))\nrows = []\n## iterating over the number of studies\nfor study_id in test_studies:\n    for condition in conditions:\n        row_id = f\"{study_id}_{condition}\"\n        row_data = {\"row_id\": row_id}\n        for label_id, label in enumerate(labels):\n            row_data[label] = value_counts[condition][label_id]\n        rows.append(row_data)\nsubmission_df = pd.DataFrame(rows)\nsubmission_df\nsubmission_df.to_csv(\"submission.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2024-08-26T09:26:35.466054Z","iopub.execute_input":"2024-08-26T09:26:35.466559Z","iopub.status.idle":"2024-08-26T09:26:35.481947Z","shell.execute_reply.started":"2024-08-26T09:26:35.466517Z","shell.execute_reply":"2024-08-26T09:26:35.480302Z"},"trusted":true},"execution_count":null,"outputs":[]}]}